{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\nfrom sklearn.svm import SVC\nfrom sklearn.cluster import KMeans\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import  RandomForestClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nfrom sklearn.metrics import r2_score,classification_report,consensus_score,completeness_score\nfrom sklearn.naive_bayes import GaussianNB\nimport plotly.animation as animation\nimport plotly.express as px\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"979b9f7c-ced4-48a6-b6f7-64b881df6bbe","_cell_guid":"b74e7c3f-fc80-4226-a91f-5c7a83099d56","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:35:17.499592Z","iopub.execute_input":"2024-02-12T14:35:17.499874Z","iopub.status.idle":"2024-02-12T14:35:17.506316Z","shell.execute_reply.started":"2024-02-12T14:35:17.499833Z","shell.execute_reply":"2024-02-12T14:35:17.505461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Importing Libraries:\n\npandas: Data manipulation and analysis.\nnumpy: Numerical operations on arrays.\nseaborn: Data visualization based on Matplotlib, providing a high-level interface.\nmatplotlib.pyplot: A 2D plotting library for creating static, animated, and interactive visualizations.\ntrain_test_split: Function for splitting datasets into training and testing sets.\nLinearRegression: Linear regression model from scikit-learn.\nSVC: Support Vector Classification model from scikit-learn.\nKMeans: K-Means clustering algorithm from scikit-learn.\nDecisionTreeClassifier: Decision Tree classifier from scikit-learn.\nRandomForestClassifier: Random Forest classifier from scikit-learn.\nGradientBoostingClassifier: Gradient Boosting classifier from scikit-learn.\nr2_score, classification_report, consensus_score, completeness_score: Evaluation metrics from scikit-learn.\nGaussianNB: Gaussian Naive Bayes classifier from scikit-learn.\nplotly.animation: Animation module from Plotly.\nplotly.express: High-level interface for creating various visualizations with Plotly.\nFiltering Warnings:\n\nThe line warnings.filterwarnings(\"ignore\") suppresses warning messages to avoid cluttering the output.","metadata":{"_uuid":"da468040-ad56-4099-9f48-fe2b9a8433c8","_cell_guid":"1e858083-6a03-4d74-affa-b49a0a7dea6d","trusted":true}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"_uuid":"4a25c6f0-ef35-4847-8a75-7be60d9bcc97","_cell_guid":"a5122709-eff8-4fdb-bfd8-cffcf44c940c","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:56.507570Z","iopub.execute_input":"2024-02-12T14:23:56.508085Z","iopub.status.idle":"2024-02-12T14:23:56.761884Z","shell.execute_reply.started":"2024-02-12T14:23:56.508057Z","shell.execute_reply":"2024-02-12T14:23:56.760762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"pd: An alias for the pandas library.\nread_csv: A function from pandas used to read data from a CSV file.\n\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\": The file path of the CSV file you are reading.\nSo, this line of code reads the contents of the CSV file located at the specified path and stores it in a DataFrame named df. This DataFrame likely contains the training data for a machine learning model.","metadata":{"_uuid":"117f7927-d96d-4ced-9732-2b3e0d9dee13","_cell_guid":"625dd452-cfc2-4483-8aa8-e8dfa2b015de","trusted":true}},{"cell_type":"code","source":"df","metadata":{"_uuid":"cd44fe73-56b4-47d8-a3f5-54934c4916dd","_cell_guid":"e3444320-f5ec-4cc3-a830-4aba058eee8a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:56.763379Z","iopub.execute_input":"2024-02-12T14:23:56.763781Z","iopub.status.idle":"2024-02-12T14:23:56.798038Z","shell.execute_reply.started":"2024-02-12T14:23:56.763746Z","shell.execute_reply":"2024-02-12T14:23:56.797032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.keys()","metadata":{"_uuid":"3626448f-cf26-4873-bace-17eda8a2714d","_cell_guid":"d0fd8969-deb9-47bc-b2c7-797a2f1c241f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:56.800182Z","iopub.execute_input":"2024-02-12T14:23:56.800535Z","iopub.status.idle":"2024-02-12T14:23:56.807414Z","shell.execute_reply.started":"2024-02-12T14:23:56.800508Z","shell.execute_reply":"2024-02-12T14:23:56.806442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"_uuid":"32db327b-16ee-48ab-a514-dbfdf5e0b044","_cell_guid":"75d21cba-cee5-461d-8236-74f7b2ca5da3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:56.808664Z","iopub.execute_input":"2024-02-12T14:23:56.808956Z","iopub.status.idle":"2024-02-12T14:23:56.896968Z","shell.execute_reply.started":"2024-02-12T14:23:56.808931Z","shell.execute_reply":"2024-02-12T14:23:56.896007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.interpolate()","metadata":{"_uuid":"598db69f-17d0-474e-aa33-2ae5c3ecdde0","_cell_guid":"5ef0f29d-4e3e-4ee5-bdf5-9191f510b770","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:56.898184Z","iopub.execute_input":"2024-02-12T14:23:56.898517Z","iopub.status.idle":"2024-02-12T14:23:56.924857Z","shell.execute_reply.started":"2024-02-12T14:23:56.898489Z","shell.execute_reply":"2024-02-12T14:23:56.923733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.value_counts()","metadata":{"_uuid":"5f20b7cf-ed9c-4cf8-ac5f-5b09df194065","_cell_guid":"b2f0d67a-4061-42df-8df3-052a767f008e","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:56.926281Z","iopub.execute_input":"2024-02-12T14:23:56.926668Z","iopub.status.idle":"2024-02-12T14:23:57.060636Z","shell.execute_reply.started":"2024-02-12T14:23:56.926632Z","shell.execute_reply":"2024-02-12T14:23:57.059616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.unstack()","metadata":{"_uuid":"f4fbfe9e-f7eb-4f92-922f-b19f3d876fc5","_cell_guid":"ee35ff9f-c4e0-4f16-bb46-59ce477e86a9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:23:57.061725Z","iopub.execute_input":"2024-02-12T14:23:57.062034Z","iopub.status.idle":"2024-02-12T14:24:47.339285Z","shell.execute_reply.started":"2024-02-12T14:23:57.062008Z","shell.execute_reply":"2024-02-12T14:24:47.338214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"_uuid":"f6a9cd1f-8efb-40b0-ac7d-62273ae2dda9","_cell_guid":"df3fa024-0cb0-47ca-a7c8-2a482edbc448","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:47.340554Z","iopub.execute_input":"2024-02-12T14:24:47.340864Z","iopub.status.idle":"2024-02-12T14:24:47.370858Z","shell.execute_reply.started":"2024-02-12T14:24:47.340838Z","shell.execute_reply":"2024-02-12T14:24:47.369813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"_uuid":"a94aaee1-b5a0-4f14-af6f-4b3ba99828c0","_cell_guid":"71567df4-d86b-4bcc-9268-7a08f84177d0","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:47.375095Z","iopub.execute_input":"2024-02-12T14:24:47.375445Z","iopub.status.idle":"2024-02-12T14:24:47.396389Z","shell.execute_reply.started":"2024-02-12T14:24:47.375414Z","shell.execute_reply":"2024-02-12T14:24:47.395290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **data visualization by using Matplotlib**","metadata":{"_uuid":"3a0b5115-5093-4911-93e3-157dc2619c88","_cell_guid":"191385ee-3d81-4359-a5ed-06db42e374e5","trusted":true}},{"cell_type":"markdown","source":"# **Distribution Graph**","metadata":{"_uuid":"2d2a602a-27e8-444c-a720-f75c2cb07b8f","_cell_guid":"16c4a300-b97f-4594-a62b-6ae482ab329f","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_id'].plot(kind='hist', bins=20, title='eeg_id')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"62170485-1f58-44d4-aca5-a1d00985e3cb","_cell_guid":"6613bb4c-2459-4c60-b0fc-41a1694ed74d","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:47.397726Z","iopub.execute_input":"2024-02-12T14:24:47.398075Z","iopub.status.idle":"2024-02-12T14:24:47.712306Z","shell.execute_reply.started":"2024-02-12T14:24:47.398028Z","shell.execute_reply":"2024-02-12T14:24:47.711242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Explanation of code:\nThe code first imports the pyplot module from the matplotlib library and aliases it as plt.\nIt then uses the plot method on the 'eeg_id' column of the DataFrame (df) to create a histogram. The kind='hist' parameter specifies that a histogram should be plotted.\nThe bins=20 parameter sets the number of bins (intervals) in the histogram to 20.\nThe title='eeg_id' parameter sets the title of the plot to 'eeg_id'.\nThe last line hides the spines on the top and right sides of the plot for better visualization.\nInterpreting the Histogram:\n\nThe 'eeg_id' is likely an identifier for EEG recordings, and the histogram provides insights into the distribution of these recordings.\nEach bin in the histogram represents a range of 'eeg_id' values, and the height of each bin indicates the frequency (count) of recordings falling within that range.\nUsefulness in EEG Analysis:\n\nDistribution Analysis: The histogram helps visualize how the 'eeg_id' values are distributed. Unusual patterns or gaps in the distribution may suggest issues or patterns in the data.\nData Integrity Check: A smooth and evenly distributed histogram may indicate good data integrity and consistent recording practices.\nIdentification of Outliers: Spikes or outliers in the histogram may point to specific recordings that require further investigation.\nTemporal Patterns: Depending on the nature of the 'eeg_id' values, the histogram might reveal temporal patterns or recording frequency over time.\nEnhancing Visualization:\n\nThe removal of spines on the top and right sides improves the clarity of the plot, making it more aesthetically pleasing.\nIn summary, the histogram of 'eeg_id' can be a valuable tool for initial exploratory data analysis in EEG research. It provides a quick overview of the distribution of EEG recordings, which can aid in identifying patterns, outliers, and potential issues in the dataset.","metadata":{"_uuid":"319ad640-84d2-40ef-87df-29ecab3b47b9","_cell_guid":"af896d47-fd51-4732-b7d2-b9de9abea058","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_sub_id'].plot(kind='hist', bins=20, title='eeg_sub_id')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"54272d24-5ee5-4bde-b7a2-a6a5bd0369d0","_cell_guid":"576d5cd9-7829-4b1a-9c27-5e1e5442f110","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:47.713524Z","iopub.execute_input":"2024-02-12T14:24:47.713790Z","iopub.status.idle":"2024-02-12T14:24:47.966941Z","shell.execute_reply.started":"2024-02-12T14:24:47.713767Z","shell.execute_reply":"2024-02-12T14:24:47.965990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Explanation of code:\nThe code imports the pyplot module from the matplotlib library and aliases it as plt.\nIt then uses the plot method on the 'eeg_sub_id' column of the DataFrame (df) to create a histogram. The kind='hist' parameter specifies that a histogram should be plotted.\nThe bins=20 parameter sets the number of bins (intervals) in the histogram to 20.\nThe title='eeg_sub_id' parameter sets the title of the plot to 'eeg_sub_id'.\nThe last line hides the spines on the top and right sides of the plot for better visualization.\nInterpreting the Histogram:\n\nThe 'eeg_sub_id' is likely a sub-identifier within the 'eeg_id' column, possibly representing different segments or sub-divisions within each EEG recording.\nThe histogram shows the distribution of 'eeg_sub_id' values, where each bin represents a range of sub-identifier values, and the height of each bin indicates the frequency (count) of occurrences within that range.\nUsefulness in EEG Analysis:\n\nSegmentation Analysis: The histogram helps visualize how different segments or sub-divisions are distributed within the EEG recordings.\nIdentifying Common Sub-structures: Patterns or clusters in the histogram may indicate common sub-structures within EEG recordings, which could be relevant for analysis.\nData Quality Check: Irregularities or unexpected patterns in the histogram may signal issues with the data or recording process.\nTemporal Patterns in Segments: If 'eeg_sub_id' has a temporal component, the histogram may reveal patterns in the distribution of segments over time.\nEnhancing Visualization:\n\nThe removal of spines on the top and right sides improves the clarity of the plot, making it more visually appealing.\nIn summary, the histogram of 'eeg_sub_id' provides insights into the distribution of sub-identifiers within EEG recordings. This information can be valuable for understanding the structure of the data, identifying patterns, and conducting quality checks on the EEG dataset.","metadata":{"_uuid":"ef4937d2-bb58-4429-91f0-a49982680b70","_cell_guid":"886b6567-fea6-4e3d-960f-309ba9930487","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_label_offset_seconds'].plot(kind='hist', bins=20, title='eeg_label_offset_seconds')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"53f6959b-b33f-4823-a5fc-d996922f0a09","_cell_guid":"b2eb0753-60bb-4cba-a8f5-06459bbc5dc9","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:47.968162Z","iopub.execute_input":"2024-02-12T14:24:47.968445Z","iopub.status.idle":"2024-02-12T14:24:48.226399Z","shell.execute_reply.started":"2024-02-12T14:24:47.968421Z","shell.execute_reply":"2024-02-12T14:24:48.225356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The given code uses Matplotlib to create a histogram of a column named 'eeg_label_offset_seconds' from a DataFrame (df). It specifies 20 bins for the histogram and sets the title to 'eeg_label_offset_seconds'. Additionally, it removes the top and right spines of the plot for better aesthetics.","metadata":{"_uuid":"74941a48-a005-498e-b1bb-73bb6ac9ac23","_cell_guid":"8a60d9bb-2e68-4f2c-af72-244a4f79ffda","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['spectrogram_id'].plot(kind='hist', bins=20, title='spectrogram_id')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"19bbc958-7925-46da-b52d-3cdc76f26225","_cell_guid":"81534349-60fb-4c2e-b56b-0aac8c923d8a","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:48.227485Z","iopub.execute_input":"2024-02-12T14:24:48.227763Z","iopub.status.idle":"2024-02-12T14:24:48.489186Z","shell.execute_reply.started":"2024-02-12T14:24:48.227738Z","shell.execute_reply":"2024-02-12T14:24:48.488278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code uses Matplotlib to generate a histogram for the 'spectrogram_id' column from the DataFrame (df). It specifies 20 bins for the histogram and sets the title as 'spectrogram_id'. The last line of code removes the top and right spines of the plot for a cleaner appearance. The 'spectrogram_id' column likely contains numerical data, and the histogram visualizes the distribution of these values within the specified bins.","metadata":{"_uuid":"81cbaf56-b55c-4c73-bab8-f9f551acc608","_cell_guid":"53989ee8-e118-4ab3-9937-f659c7ae7e19","trusted":true}},{"cell_type":"markdown","source":"# **Categorical distributions**","metadata":{"_uuid":"965a78a2-cbcf-476f-8ae9-2af04f11db13","_cell_guid":"e2b1fc76-875a-4d1a-b077-32f2db39e0ef","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\ndf.groupby('expert_consensus').size().plot(kind='barh', color=sns.palettes.mpl_palette('Dark2'))\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"70cfc27b-0011-47dc-9d15-b81c7f718a70","_cell_guid":"19838de9-b4f8-45fb-befd-7724da1f9c08","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:48.490521Z","iopub.execute_input":"2024-02-12T14:24:48.490806Z","iopub.status.idle":"2024-02-12T14:24:48.704556Z","shell.execute_reply.started":"2024-02-12T14:24:48.490780Z","shell.execute_reply":"2024-02-12T14:24:48.703766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code uses Matplotlib and Seaborn to create a horizontal bar chart representing the count of occurrences for each unique value in the 'expert_consensus' column of the DataFrame (df). The groupby('expert_consensus').size() part groups the data by the 'expert_consensus' column and calculates the size (count) of each group.\n\nThe plot(kind='barh', color=sns.palettes.mpl_palette('Dark2')) line creates a horizontal bar chart with colors from the 'Dark2' palette of Seaborn. Each bar represents a unique value in the 'expert_consensus' column, and its length corresponds to the count of occurrences.","metadata":{"_uuid":"04caa7c2-6607-4de3-a9d4-7881180030be","_cell_guid":"1848d0db-a74d-4ac6-93ce-d4d963c611e0","trusted":true}},{"cell_type":"markdown","source":"# **2-D distributions Plot Visualization**","metadata":{"_uuid":"f50fe16a-677d-418d-a3eb-f2a7bef4c2b0","_cell_guid":"97346902-f5e8-44a5-8df1-a3c598db408b","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf.plot(kind='scatter', x='eeg_id', y='eeg_sub_id', s=32, alpha=.8)\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"931117d4-bf6e-41c3-9123-fbacd97ee8d7","_cell_guid":"19f2000a-f5b4-450c-ac7f-c46b2baaffd7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:48.705678Z","iopub.execute_input":"2024-02-12T14:24:48.705964Z","iopub.status.idle":"2024-02-12T14:24:49.386243Z","shell.execute_reply.started":"2024-02-12T14:24:48.705939Z","shell.execute_reply":"2024-02-12T14:24:49.385271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{"_uuid":"d16960ba-d0f1-403d-b1d9-2ef106154b80","_cell_guid":"245fddd7-1e6b-499d-b1cc-2d205441b466","trusted":true}},{"cell_type":"markdown","source":"The provided code uses Matplotlib to create a scatter plot using the 'eeg_id' column as the x-axis and the 'eeg_sub_id' column as the y-axis from the DataFrame (df). Here's a breakdown of the code:\n\nScatter Plot Generation:\n\ndf.plot(kind='scatter', x='eeg_id', y='eeg_sub_id', s=32, alpha=.8): This line of code generates a scatter plot using the 'eeg_id' as the x-axis and 'eeg_sub_id' as the y-axis. The s=32 parameter sets the marker size to 32, and alpha=.8 controls the transparency of the markers, making them slightly transparent.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot, improving its visual appeal.\nIn the context of brain activity data, the scatter plot visualizes the relationship between 'eeg_id' and 'eeg_sub_id'. Each point on the plot represents a pair of values from these two columns, where 'eeg_id' is on the x-axis and 'eeg_sub_id' is on the y-axis. The marker size and transparency add extra visual information to the plot.","metadata":{"_uuid":"5756065f-8dc7-48c0-b369-b5baf160c29d","_cell_guid":"61fa1722-a750-431a-ae0a-949aaa37d537","trusted":true}},{"cell_type":"markdown","source":"","metadata":{"_uuid":"ee10c1fa-5fe0-4670-a020-3df38d995b4b","_cell_guid":"5b01f0da-ae90-414a-9c80-e8602fcb295b","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf.plot(kind='scatter', x='eeg_sub_id', y='eeg_label_offset_seconds', s=32, alpha=.8)\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"0fd4cad8-8c91-49a3-8f3f-085413695587","_cell_guid":"6448a1a4-c04d-47ed-9216-9ad0fc7af3f4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:49.387366Z","iopub.execute_input":"2024-02-12T14:24:49.387636Z","iopub.status.idle":"2024-02-12T14:24:50.064449Z","shell.execute_reply.started":"2024-02-12T14:24:49.387612Z","shell.execute_reply":"2024-02-12T14:24:50.063557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The provided code uses Matplotlib to create a scatter plot using the 'eeg_id' column as the x-axis and the 'eeg_sub_id' column as the y-axis from the DataFrame (df). Here's a breakdown of the code:\n\nScatter Plot Generation:\n\ndf.plot(kind='scatter', x='eeg_id', y='eeg_sub_id', s=32, alpha=.8): This line of code generates a scatter plot using the 'eeg_id' as the x-axis and 'eeg_sub_id' as the y-axis. The s=32 parameter sets the marker size to 32, and alpha=.8 controls the transparency of the markers, making them slightly transparent.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot, improving its visual appeal.\nIn the context of brain activity data, the scatter plot visualizes the relationship between 'eeg_id' and 'eeg_sub_id'. Each point on the plot represents a pair of values from these two columns, where 'eeg_id' is on the x-axis and 'eeg_sub_id' is on the y-axis. The marker size and transparency add extra visual information to the plot.","metadata":{"_uuid":"2c7e972d-f886-4418-bd35-716e43dad6d7","_cell_guid":"6c0ffb21-f6dd-48be-aa95-6333f94ea3e5","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf.plot(kind='scatter', x='eeg_label_offset_seconds', y='spectrogram_id', s=32, alpha=.8)\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"2b29054e-51fa-4a30-b7e0-d652ba31da84","_cell_guid":"35f7bc43-a995-4204-922f-5a2d4266e554","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:50.065785Z","iopub.execute_input":"2024-02-12T14:24:50.066405Z","iopub.status.idle":"2024-02-12T14:24:50.724688Z","shell.execute_reply.started":"2024-02-12T14:24:50.066362Z","shell.execute_reply":"2024-02-12T14:24:50.723761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code creates a scatter plot using Matplotlib, where 'eeg_label_offset_seconds' is plotted on the x-axis and 'spectrogram_id' on the y-axis from the DataFrame (df). Here's an explanation:\n\nScatter Plot Generation:\n\ndf.plot(kind='scatter', x='eeg_label_offset_seconds', y='spectrogram_id', s=32, alpha=.8): This line of code generates a scatter plot with 'eeg_label_offset_seconds' on the x-axis and 'spectrogram_id' on the y-axis. The s=32 parameter sets the marker size to 32, and alpha=.8 controls the transparency of the markers.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance.\nIn the context of your dataset, the scatter plot visualizes the relationship between 'eeg_label_offset_seconds' and 'spectrogram_id'. Each point on the plot represents a pair of values from these two columns. The marker size and transparency add visual details to the plot.","metadata":{"_uuid":"89fbded2-815c-4c40-aba0-4ab91a4a15c4","_cell_guid":"b929535f-d4a5-48e9-9af2-9b5f3538d4ed","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf.plot(kind='scatter', x='spectrogram_id', y='spectrogram_sub_id', s=32, alpha=.8)\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"ac3b458e-d1ab-424f-b3e9-903237990b9e","_cell_guid":"8dc9afad-dc22-4e52-9348-e99664cc4b13","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:50.726003Z","iopub.execute_input":"2024-02-12T14:24:50.726613Z","iopub.status.idle":"2024-02-12T14:24:51.382679Z","shell.execute_reply.started":"2024-02-12T14:24:50.726578Z","shell.execute_reply":"2024-02-12T14:24:51.381796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"he provided code generates a scatter plot using Matplotlib, where 'spectrogram_id' is plotted on the x-axis and 'spectrogram_sub_id' on the y-axis from the DataFrame (df). Here's an explanation:\n\nScatter Plot Generation:\n\ndf.plot(kind='scatter', x='spectrogram_id', y='spectrogram_sub_id', s=32, alpha=.8): This line of code creates a scatter plot with 'spectrogram_id' on the x-axis and 'spectrogram_sub_id' on the y-axis. The s=32 parameter sets the marker size to 32, and alpha=.8 controls the transparency of the markers.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance.\nRegarding brain activity analysis, the interpretation depends on the specific meaning of 'spectrogram_id' and 'spectrogram_sub_id' in your dataset:\n\n'spectrogram_id': If this represents different spectrograms or patterns of brain activity, then the scatter plot could help visualize the relationship between different spectrogram IDs.\n\n'spectrogram_sub_id': If this represents sub-IDs or details related to the spectrograms, the scatter plot may show how these sub-IDs vary with respect to different spectrogram IDs.","metadata":{"_uuid":"da84b6e8-56ec-48f6-955c-c6201a07ea16","_cell_guid":"2cae7632-840a-4f24-a2f0-256546e26603","trusted":true}},{"cell_type":"markdown","source":"# **Time Series Plot Visualization**","metadata":{"_uuid":"0a939dde-f1b3-4c80-ae5f-a2ca762a8331","_cell_guid":"b40ab27b-d2e4-4fb3-9e75-255df248b1be","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\ndef _plot_series(series, series_name, series_index=0):\n  from matplotlib import pyplot as plt\n  import seaborn as sns\n  palette = list(sns.palettes.mpl_palette('Dark2'))\n  xs = series['spectrogram_id']\n  ys = series['eeg_id']\n\n  plt.plot(xs, ys, label=series_name, color=palette[series_index % len(palette)])\n\nfig, ax = plt.subplots(figsize=(10, 5.2), layout='constrained')\ndf_sorted = df.sort_values('spectrogram_id', ascending=True)\nfor i, (series_name, series) in enumerate(df_sorted.groupby('expert_consensus')):\n  _plot_series(series, series_name, i)\n  fig.legend(title='expert_consensus', bbox_to_anchor=(1, 1), loc='upper left')\nsns.despine(fig=fig, ax=ax)\nplt.xlabel('spectrogram_id')\n_ = plt.ylabel('eeg_id')","metadata":{"_uuid":"58425fb0-ce5d-4978-81f7-9b74a1dde800","_cell_guid":"0740a4e4-fc91-4c6f-8768-f203bee5cf36","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2024-02-12T14:24:51.383773Z","iopub.execute_input":"2024-02-12T14:24:51.384048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code defines a function _plot_series to visualize time series data in a line plot. It then creates a subplot using Matplotlib and Seaborn, sorting the DataFrame by 'spectrogram_id'. For each group of data grouped by 'expert_consensus', it calls the _plot_series function to plot the 'spectrogram_id' against 'eeg_id' with different line colors. Finally, it adjusts the legend, removes spines, and labels the axes.","metadata":{"_uuid":"1e3fa207-67ef-4d38-b066-39c79f4c4069","_cell_guid":"2663979d-8a5a-4151-88e7-d75f8d1e6039","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\ndef _plot_series(series, series_name, series_index=0):\n  from matplotlib import pyplot as plt\n  import seaborn as sns\n  palette = list(sns.palettes.mpl_palette('Dark2'))\n  xs = series['spectrogram_id']\n  ys = series['eeg_sub_id']\n\n  plt.plot(xs, ys, label=series_name, color=palette[series_index % len(palette)])\n\nfig, ax = plt.subplots(figsize=(10, 5.2), layout='constrained')\ndf_sorted = df.sort_values('spectrogram_id', ascending=True)\nfor i, (series_name, series) in enumerate(df_sorted.groupby('expert_consensus')):\n  _plot_series(series, series_name, i)\n  fig.legend(title='expert_consensus', bbox_to_anchor=(1, 1), loc='upper left')\nsns.despine(fig=fig, ax=ax)\nplt.xlabel('spectrogram_id')\n_ = plt.ylabel('eeg_sub_id')","metadata":{"_uuid":"0e152558-6b75-42ca-a496-44c71ad90068","_cell_guid":"0fd4b5f5-f12b-4752-800c-03fe4bb8c8c6","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code appears to generate a line plot visualizing the relationship between 'spectrogram_id' and 'eeg_sub_id' for different groups defined by 'expert_consensus'. Here's an explanation, aligned with brain EEG analysis:\n\nFunction Definition:\n\ndef _plot_series(series, series_name, series_index=0):: This function takes a time series data (series) and its name (series_name) and plots 'spectrogram_id' against 'eeg_sub_id' using different colors for each series.\nData Plotting:\n\nxs = series['spectrogram_id']: Extracts the 'spectrogram_id' as x-values.\nys = series['eeg_sub_id']: Extracts the 'eeg_sub_id' as y-values.\nplt.plot(xs, ys, label=series_name, color=palette[series_index % len(palette)]): Plots the line for the current series with a specific color.\nSubplots Initialization:\n\nfig, ax = plt.subplots(figsize=(10, 5.2), layout='constrained'): Initializes a subplot with a specified size.\nData Sorting and Grouping:\n\ndf_sorted = df.sort_values('spectrogram_id', ascending=True): Sorts the DataFrame by 'spectrogram_id'.\nfor i, (series_name, series) in enumerate(df_sorted.groupby('expert_consensus')):: Iterates through the groups defined by 'expert_consensus' after sorting.\nPlotting Data for Each Group:\n\n_plot_series(series, series_name, i): Calls the function to plot the 'spectrogram_id' against 'eeg_sub_id' for each group.\nfig.legend(title='expert_consensus', bbox_to_anchor=(1, 1), loc='upper left'): Adds a legend with the title 'expert_consensus' to the plot.\nAesthetics Adjustment:\n\nsns.despine(fig=fig, ax=ax): Removes spines for a cleaner appearance.\nplt.xlabel('spectrogram_id'): Adds a label to the x-axis.\n_ = plt.ylabel('eeg_sub_id'): Adds a label to the y-axis.\nIn the context of brain EEG analysis, this code helps visualize how 'eeg_sub_id' varies with 'spectrogram_id' for different groups defined by 'expert_consensus'. Each line represents a specific group, and the legend helps identify which line corresponds to each 'expert_consensus' category. The plot may reveal patterns or differences in EEG sub-IDs across different spectrogram IDs and expert consensus groups","metadata":{"_uuid":"621ac52e-145a-4930-8c84-f9e3b6d155a7","_cell_guid":"be9479ae-2417-44b7-8f5d-09066d28b4a3","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\ndef _plot_series(series, series_name, series_index=0):\n  from matplotlib import pyplot as plt\n  import seaborn as sns\n  palette = list(sns.palettes.mpl_palette('Dark2'))\n  xs = series['spectrogram_id']\n  ys = series['eeg_label_offset_seconds']\n\n  plt.plot(xs, ys, label=series_name, color=palette[series_index % len(palette)])\n\nfig, ax = plt.subplots(figsize=(10, 5.2), layout='constrained')\ndf_sorted = df.sort_values('spectrogram_id', ascending=True)\nfor i, (series_name, series) in enumerate(df_sorted.groupby('expert_consensus')):\n  _plot_series(series, series_name, i)\n  fig.legend(title='expert_consensus', bbox_to_anchor=(1, 1), loc='upper left')\nsns.despine(fig=fig, ax=ax)\nplt.xlabel('spectrogram_id')\n_ = plt.ylabel('eeg_label_offset_seconds')","metadata":{"_uuid":"b034a41c-254b-4276-b1aa-c3f90960573f","_cell_guid":"2ac80903-9035-4f73-bcb4-741e70991132","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Faceted distributions visualization**","metadata":{"_uuid":"8754860e-73f0-4cec-9805-c8066d20b927","_cell_guid":"ea5e115f-4e35-4044-923f-c2517de11f0c","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\nfigsize = (12, 1.2 * len(df['expert_consensus'].unique()))\nplt.figure(figsize=figsize)\nsns.violinplot(df, x='eeg_id', y='expert_consensus', inner='box', palette='Dark2')\nsns.despine(top=True, right=True, bottom=True, left=True)","metadata":{"_uuid":"a3460b14-206c-4a97-899f-38c26cf389db","_cell_guid":"d72ab6fc-9953-4173-85aa-d91e591053ba","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code generates a violin plot using Seaborn to visualize the distribution of 'eeg_id' values across different categories of 'expert_consensus'. Here's a concise explanation:\n\nFigure Size Setup:\n\nfigsize = (12, 1.2 * len(df['expert_consensus'].unique())): Determines the figure size based on the number of unique values in the 'expert_consensus' column.\nplt.figure(figsize=figsize): Sets up the figure with the calculated size.\nViolin Plot Generation:\n\nsns.violinplot(df, x='eeg_id', y='expert_consensus', inner='box', palette='Dark2'): Creates a violin plot showing the distribution of 'eeg_id' values for each category of 'expert_consensus'. The inner box represents the quartiles, and the 'Dark2' palette provides color variations.\nAesthetics Adjustment:\n\nsns.despine(top=True, right=True, bottom=True, left=True): Removes spines for a cleaner appearance.\nIn the context of brain EEG analysis, this plot helps visualize the distribution of 'eeg_id' values across different expert consensus categories. The width of the violin plots indicates the density of data points at different 'eeg_id' values, and the inner box provides information about the quartiles within each category.","metadata":{"_uuid":"b688efd5-dc3e-4719-a3b7-e8ed94785360","_cell_guid":"c0310e9c-a08a-4a76-9c3e-3f367e032ee1","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\nimport seaborn as sns\nfigsize = (12, 1.2 * len(df['expert_consensus'].unique()))\nplt.figure(figsize=figsize)\nsns.violinplot(df, x='eeg_sub_id', y='expert_consensus', inner='box', palette='Dark2')\nsns.despine(top=True, right=True, bottom=True, left=True)","metadata":{"_uuid":"1780fb49-f4a2-48d6-8b4d-049b7461d4c0","_cell_guid":"77720960-e548-421a-b3c1-d7d35a1760fe","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Figure Size Setup:\n\nfigsize = (12, 1.2 * len(df['expert_consensus'].unique())): Determines the figure size based on the number of unique values in the 'expert_consensus' column.\nplt.figure(figsize=figsize): Sets up the figure with the calculated size.\nViolin Plot Generation:\n\nsns.violinplot(df, x='eeg_sub_id', y='expert_consensus', inner='box', palette='Dark2'): Creates a violin plot to illustrate the distribution of 'eeg_sub_id' values across different categories of 'expert_consensus'. The width of the violins depicts data density, and the inner box provides information on quartiles. The 'Dark2' palette adds color variations.\nAesthetics Adjustment:\n\nsns.despine(top=True, right=True, bottom=True, left=True): Removes spines for a cleaner appearance.\nIn the context of brain EEG analysis, this plot visually communicates how 'eeg_sub_id' values vary within different expert consensus categories. The width of the violins signifies the density of data points at different 'eeg_sub_id' values, with the inner box providing details about quartiles. The 'Dark2' palette enhances visual distinction between different categories.","metadata":{"_uuid":"a40cc0d3-e815-4b04-a8c6-5f0ce6f2f2cd","_cell_guid":"7dfd2e78-e592-42be-b154-c0c9f2cd38e9","trusted":true}},{"cell_type":"markdown","source":"# **Corroration plot visualization**","metadata":{"_uuid":"c465737d-8240-4175-b342-30f9d5d1b59c","_cell_guid":"3efcab8c-db42-44d0-a4c6-d2ed8fd8fca3","trusted":true}},{"cell_type":"markdown","source":"# Distributions plots visualization","metadata":{"_uuid":"97914a3d-9e16-4eb9-9297-5813ad052e2f","_cell_guid":"3e6fbcd2-7d62-4a3a-af77-d85817eb211f","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_id'].plot(kind='hist', bins=20, title='eeg_id')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"9c4d3f19-f4c8-4e6c-90c2-806efd7fcb17","_cell_guid":"5354406d-f14b-4f04-8e0b-df21f18141f4","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"his code uses Matplotlib to create a histogram for the 'eeg_id' column from the DataFrame (df). Here's a breakdown of the code:\n\nHistogram Generation:\n\ndf['eeg_id'].plot(kind='hist', bins=20, title='eeg_id'): This line of code generates a histogram for the 'eeg_id' column. It specifies 20 bins for the histogram using the bins=20 parameter. The kind='hist' indicates that it's a histogram plot.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It accesses the current Axes (plt.gca()) and modifies the visibility of the specified spines.\nIn summary, the code produces a histogram that visualizes the distribution of values in the 'eeg_id' column. The histogram is divided into 20 bins, providing insights into how the 'eeg_id' values are distributed in the dataset. The removal of the top and right spines enhances the clarity of the plot.","metadata":{"_uuid":"b8dd03b8-22bd-4e7d-a09c-aa082a1062fd","_cell_guid":"fe70bf46-ac08-4ba3-b9f8-92aff9bab667","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_sub_id'].plot(kind='hist', bins=20, title='eeg_sub_id')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"b81480d8-5906-4f8e-849c-068702b6ebc4","_cell_guid":"d27c44c1-8a05-4a31-9956-3d804207144a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code uses Matplotlib to create a histogram for the 'eeg_sub_id' column from the DataFrame (df). Here's an explanation of the code:\n\nHistogram Generation:\n\ndf['eeg_sub_id'].plot(kind='hist', bins=20, title='eeg_sub_id'): This line of code generates a histogram for the 'eeg_sub_id' column. It specifies 20 bins for the histogram using the bins=20 parameter. The kind='hist' indicates that it's a histogram plot.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It accesses the current Axes (plt.gca()) and modifies the visibility of the specified spines.\nIn summary, the code produces a histogram that visualizes the distribution of values in the 'eeg_sub_id' column. The histogram is divided into 20 bins, providing insights into how the 'eeg_sub_id' values are distributed in the dataset. The removal of the top and right spines enhances the clarity of the plot, similar to the previous explanation for 'eeg_id'.","metadata":{"_uuid":"721ed906-33fa-4c95-b979-a3d631fc72ee","_cell_guid":"a778d5fc-259c-42b2-a381-bc09bc0115db","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_label_offset_seconds'].plot(kind='hist', bins=20, title='eeg_label_offset_seconds')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"0368772a-f110-4635-9010-d882ac07ecdc","_cell_guid":"60ef65f4-86c6-472a-8f77-7c47b9f67acc","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code generates a histogram for the 'eeg_label_offset_seconds' column from the DataFrame (df). Here's a breakdown of the code, aligned with the context of the above dataset:\n\nHistogram Generation:\n\ndf['eeg_label_offset_seconds'].plot(kind='hist', bins=20, title='eeg_label_offset_seconds'): This line of code generates a histogram for the 'eeg_label_offset_seconds' column. It specifies 20 bins for the histogram using the bins=20 parameter. The kind='hist' indicates that it's a histogram plot.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It accesses the current Axes (plt.gca()) and modifies the visibility of the specified spines.\nIn the context of the provided dataset, 'eeg_label_offset_seconds' likely represents a temporal offset associated with EEG labels. The histogram gives you an overview of the distribution of these offset values. The number of bins (20 in this case) divides the range of values into intervals, allowing you to see how frequently different offset values occur.","metadata":{"_uuid":"005a8a2a-12d9-425a-8c45-62ae8b0a9a20","_cell_guid":"6da66c90-757a-477a-8f0f-c73a269e402c","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['spectrogram_id'].plot(kind='hist', bins=20, title='spectrogram_id')\nplt.gca().spines[['top', 'right',]].set_visible(False)","metadata":{"_uuid":"c07aaa21-fe66-4d8f-8d17-5e12d186d2a7","_cell_guid":"5116f283-ec9f-42f2-b3e4-7d98c8de0831","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code generates a histogram for the 'spectrogram_id' column from the DataFrame (df). Here's an explanation of the code:\n\nHistogram Generation:\n\ndf['spectrogram_id'].plot(kind='hist', bins=20, title='spectrogram_id'): This line of code generates a histogram for the 'spectrogram_id' column. It specifies 20 bins for the histogram using the bins=20 parameter. The kind='hist' indicates that it's a histogram plot.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right',]].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It accesses the current Axes (plt.gca()) and modifies the visibility of the specified spines.\nIn the context of the dataset, 'spectrogram_id' likely represents identifiers for different spectrograms associated with brain activity. The histogram provides a visual representation of the distribution of 'spectrogram_id' values. The number of bins (20 in this case) helps you observe the frequency of different spectrogram IDs in the dataset.","metadata":{"_uuid":"88e8bcff-2a63-4372-b18a-3718779fec6f","_cell_guid":"ddf3f4d3-3ff2-4cc5-8564-85becf9ec0a3","trusted":true}},{"cell_type":"markdown","source":"# **Values visualization**","metadata":{"_uuid":"2ac290ec-f1f1-49a4-809b-c488dabcdde7","_cell_guid":"c3f46d54-e104-435e-970a-3373ee8543bc","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_id'].plot(kind='line', figsize=(8, 4), title='eeg_id')\nplt.gca().spines[['top', 'right']].set_visible(False)","metadata":{"_uuid":"897637d8-36cf-4937-9383-faf0a4acd6eb","_cell_guid":"2a38c3d2-5431-4f0a-a314-ff457a0f2780","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code generates a line plot for the 'eeg_id' column from the DataFrame (df). Here's an explanation of the code:\n\nLine Plot Generation:\n\ndf['eeg_id'].plot(kind='line', figsize=(8, 4), title='eeg_id'): This line of code creates a line plot for the 'eeg_id' column. It uses kind='line' to specify the plot type, and figsize=(8, 4) sets the figure size. The title of the plot is set to 'eeg_id'.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right']].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It accesses the current Axes (plt.gca()) and modifies the visibility of the specified spines.\nIn the context of the dataset, 'eeg_id' likely represents identifiers associated with EEG data. The line plot shows the trend or pattern of 'eeg_id' values over the dataset. Each point on the line represents a specific 'eeg_id' value, and the line connects these points in sequential order.","metadata":{"_uuid":"86207271-898f-4487-87ab-c3af294feb2d","_cell_guid":"b0ca3c6a-9aac-4fc9-b1e5-4590b609f838","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_sub_id'].plot(kind='line', figsize=(8, 4), title='eeg_sub_id')\nplt.gca().spines[['top', 'right']].set_visible(False)","metadata":{"_uuid":"b6bc904c-a4cb-433c-abe0-119f430bb28d","_cell_guid":"b2edfd5b-2e3f-435b-8dd1-953fce093a7f","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code generates a line plot for the 'eeg_sub_id' column from the DataFrame (df). Here's an alternative explanation:\n\nLine Plot Generation:\n\ndf['eeg_sub_id'].plot(kind='line', figsize=(8, 4), title='eeg_sub_id'): This line of code creates a line plot to visualize the variations in 'eeg_sub_id' over the dataset. It uses kind='line' to specify the plot type, and figsize=(8, 4) sets the figure size. The title of the plot is set to 'eeg_sub_id'.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right']].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It enhances the focus on the essential content of the plot.\nIn the context of the dataset, 'eeg_sub_id' likely represents identifiers associated with subsets of EEG data. The line plot visually presents the changes or patterns in 'eeg_sub_id' values throughout the dataset. Each point on the line corresponds to a specific 'eeg_sub_id', and the line connects these points sequentially.\n\nThe removal of the top and right spines contributes to a neat and uncluttered appearance, allowing you to concentrate on the trends or variations in 'eeg_sub_id'. This type of visualization is beneficial for understanding how 'eeg_sub_id' evolves or fluctuates over the dataset.","metadata":{"_uuid":"2e031db7-83ab-4168-9495-084daafd0395","_cell_guid":"a3ac23ed-a18b-4d9f-893e-32c3807ff7eb","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['eeg_label_offset_seconds'].plot(kind='line', figsize=(8, 4), title='eeg_label_offset_seconds')\nplt.gca().spines[['top', 'right']].set_visible(False)","metadata":{"_uuid":"931991d1-c941-4c22-ad71-7e7b229a8304","_cell_guid":"262c8478-79aa-4cda-909f-ca72dd25b9f5","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This code generates a line plot for the 'eeg_label_offset_seconds' column from the DataFrame (df). Let's break down the code:\n\nLine Plot Generation:\n\ndf['eeg_label_offset_seconds'].plot(kind='line', figsize=(8, 4), title='eeg_label_offset_seconds'): This line of code creates a line plot to visualize the changes in 'eeg_label_offset_seconds' over the dataset. It uses kind='line' to specify the plot type, and figsize=(8, 4) sets the figure size. The title of the plot is set to 'eeg_label_offset_seconds'.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right']].set_visible(False): This line removes the top and right spines of the plot for a cleaner appearance. It improves the focus on the primary content of the plot.\nIn the context of the dataset, 'eeg_label_offset_seconds' likely represents temporal offsets associated with EEG labels. The line plot visually represents how these offset values change throughout the dataset. Each point on the line corresponds to a specific time offset, and the line connects these points sequentially.\n\nThe removal of the top and right spines enhances the clarity of the plot and reduces unnecessary visual elements. This type of visualization is useful for understanding trends or patterns in 'eeg_label_offset_seconds' over the dataset.","metadata":{"_uuid":"b3a3a022-48f4-4106-9093-7aa490237932","_cell_guid":"5f89fe98-064e-4201-b121-e737e80f653f","trusted":true}},{"cell_type":"code","source":"from matplotlib import pyplot as plt\ndf['spectrogram_id'].plot(kind='line', figsize=(8, 4), title='spectrogram_id')\nplt.gca().spines[['top', 'right']].set_visible(False)","metadata":{"_uuid":"1b9cbc73-2d50-4537-8fc4-ad6ccc8ff52f","_cell_guid":"59641a9a-ca89-4762-acbe-e8462d63ecee","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Line Plot Generation:\n\ndf['spectrogram_id'].plot(kind='line', figsize=(8, 4), title='spectrogram_id'): This line of code generates a line plot, illustrating the patterns and changes in the 'spectrogram_id' values across the dataset. The plot type is specified as a line plot (kind='line'), and the figure size is set to (8, 4). The title of the plot is 'spectrogram_id'.\nAesthetics Adjustment:\n\nplt.gca().spines[['top', 'right']].set_visible(False): This line removes the top and right spines of the plot for a more streamlined appearance. It accesses the current Axes (plt.gca()) and adjusts the visibility of the specified spines.\nIn the context of the dataset, 'spectrogram_id' likely serves as unique identifiers for different spectrograms related to brain activity. The line plot visually traces the progression of 'spectrogram_id' values, revealing any trends or shifts over the dataset. Each point on the line corresponds to a specific 'spectrogram_id', and the connecting line helps visualize the sequential changes.\n\nBy eliminating the top and right spines, the plot is simplified and more focused, allowing you to better appreciate the temporal evolution or patterns in 'spectrogram_id'. This type of visualization is particularly useful for observing trends in sequential data, such as the changing identifiers associated with spectrograms over time.","metadata":{"_uuid":"4d4b6874-a125-443a-8c38-13f2ba861939","_cell_guid":"c0dcdcbf-9f1f-4f0d-a2fd-176a8c4751d4","trusted":true}},{"cell_type":"markdown","source":"# **PLOTLY VISUALIZATION**","metadata":{"_uuid":"2b327fd2-a545-4c5a-aa8b-63e460f349b7","_cell_guid":"e9eef296-8a67-44d4-a30a-28c5103521d9","trusted":true}},{"cell_type":"code","source":"fig = px.area(df, x=\"eeg_id\", y=\"spectrogram_id\", color=\"seizure_vote\", line_group=\"lpd_vote\")\nfig.show()","metadata":{"_uuid":"cf54877e-8b07-498b-9559-6b4d1a3d1d16","_cell_guid":"1e30810a-62f2-45be-ab96-98f46058a896","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code you provided uses the plotly.express library to create an area plot with the following specifications:\n\nx=\"eeg_id\": The 'eeg_id' column from the DataFrame (df) is used as the x-axis values.\ny=\"spectrogram_id\": The 'spectrogram_id' column is used as the y-axis values.\ncolor=\"seizure_vote\": The color of each area is determined by the 'seizure_vote' column values.\nline_group=\"expert_consensus\": The 'expert_consensus' column is used to group data points for the lines in the area plot.\nHere's an interpretation of what this plot might convey:\n\nX-Axis (eeg_id): The x-axis represents EEG identifiers, indicating different EEG data points or instances.\n\nY-Axis (spectrogram_id): The y-axis represents spectrogram identifiers, suggesting different patterns or features in the brain activity captured by spectrograms.\n\nArea Color (seizure_vote): The color of each area is determined by the 'seizure_vote' values. This could indicate the level of agreement or voting on whether a seizure is present or not.\n\nLine Grouping (expert_consensus): The lines in the area plot are grouped based on the 'expert_consensus' values. This grouping may reveal patterns or variations in the agreement among experts regarding seizures for different EEG and spectrogram combinations.\n\nIn summary, this area plot provides a visual representation of how 'eeg_id' and 'spectrogram_id' values relate to each other, with color representing seizure voting and lines grouped by expert consensus. The plot aims to illustrate patterns or trends in the data, potentially highlighting instances of interest related to seizure voting and expert consensus.","metadata":{"_uuid":"e0d7c0d6-c207-4d8e-bd99-1ec02bf3ddfa","_cell_guid":"06a3740e-47ab-4a17-a706-5a71eaaf8d47","trusted":true}},{"cell_type":"code","source":"fig = px.funnel(df, x='spectrogram_label_offset_seconds', y='grda_vote')\nfig.show()","metadata":{"_uuid":"d2a949bc-2ac0-4bc0-8c54-398a1309bae3","_cell_guid":"00b3d755-fc22-451c-a5f9-73d8b4acdaf3","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code you provided uses the plotly.express library to create a funnel plot with the following specifications:\n\nx='spectrogram_label_offset_seconds': The x-axis values are taken from the 'spectrogram_label_offset_seconds' column in the DataFrame (df).\ny='expert_consensus': The y-axis values are taken from the 'expert_consensus' column.\nHere's an interpretation of what this funnel plot might convey:\n\nX-Axis (spectrogram_label_offset_seconds): The x-axis represents the 'spectrogram_label_offset_seconds', which likely corresponds to some time-related values associated with spectrogram labels.\n\nY-Axis (expert_consensus): The y-axis represents 'expert_consensus', indicating different categories or levels of consensus among experts.\n\nFunnel Shape: The funnel plot typically visualizes a process, where the width of the funnel at different levels represents the quantity or magnitude of data points.\n\nIn summary, this funnel plot is likely showing how the 'spectrogram_label_offset_seconds' values are distributed across different levels of 'expert_consensus'. The narrowing or widening of the funnel at different points could reveal patterns in the data, such as concentration or dispersion of 'spectrogram_label_offset_seconds' values within various consensus levels among experts.","metadata":{"_uuid":"5e412d6f-9850-4884-b00d-0abfe9a2dfc4","_cell_guid":"e8ed0cbe-2f4c-44ee-8be0-58147bcf521d","trusted":true}},{"cell_type":"code","source":"fig = px.pie(df, values='patient_id', names=\"expert_consensus\", title='The brain activity detection with respect to patients and experts  observing the harmness activity')\nfig.show()","metadata":{"_uuid":"e470e631-f1a4-47e3-ad8e-6898a1b1a963","_cell_guid":"64b4aa5b-0ad4-4965-8ca4-d7dd277ad6b0","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code you provided uses the plotly.express library to create a pie chart with the following specifications:\n\nvalues='patient_id': The values for the pie chart are taken from the 'patient_id' column in the DataFrame (df).\nnames=\"expert_consensus\": The names or categories for the pie chart slices are determined by the unique values in the 'expert_consensus' column.\ntitle='The brain activity detection with respect to patients and experts observing the harmful activity': The title of the pie chart indicates that it visualizes brain activity detection concerning patients and experts observing harmful activities.\nHere's an interpretation of what this pie chart might convey:\n\nValues (patient_id): The pie chart segments represent the distribution of 'patient_id' values, which likely correspond to different patients.\n\nNames (expert_consensus): Each segment is labeled based on the unique values in the 'expert_consensus' column, representing different levels of consensus or observations by experts regarding harmful brain activity.\n\nTitle: The title suggests that the pie chart is illustrating brain activity detection, specifically focusing on the relationship between patients and experts' observations of harmful activities.\n\nIn summary, this pie chart provides a visual overview of how brain activity detection is distributed among different patients, with each segment representing a patient and the colors or shades within each segment indicating levels of expert consensus on harmful activity observations.","metadata":{"_uuid":"b7aba753-9191-41f6-9450-d0fda5f155ba","_cell_guid":"e093ed67-0111-4170-b854-5a01bd4dbbcb","trusted":true}},{"cell_type":"code","source":"fig = px.sunburst(df, path=['lpd_vote', 'gpd_vote', 'grda_vote'], values='patient_id')\nfig.show()","metadata":{"_uuid":"4dd11047-6f15-4dc0-aecd-e3c043bdf3e1","_cell_guid":"8c658b7e-58c4-46eb-97e1-3c534909cb53","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code you provided uses the plotly.express library to create a sunburst chart with the following specifications:\n\npath=['lpd_vote', 'gpd_vote', 'grda_vote']: The hierarchical path is defined by the columns 'lpd_vote', 'gpd_vote', and 'grda_vote' from the DataFrame (df).\nvalues='patient_id': The values associated with each sunburst segment are taken from the 'patient_id' column.\nHere's an interpretation of what this sunburst chart might convey:\n\nHierarchy (lpd_vote, gpd_vote, grda_vote): The sunburst chart is structured hierarchically based on the values in the 'lpd_vote', 'gpd_vote', and 'grda_vote' columns. Each level of the hierarchy represents a subcategory, and the chart visualizes how these subcategories relate to each other.\n\nValues (patient_id): The size or weight of each sunburst segment is determined by the 'patient_id' values. This could represent the number of patients associated with each combination of 'lpd_vote', 'gpd_vote', and 'grda_vote'.\n\nIn summary, the sunburst chart provides a hierarchical visualization, illustrating how patients are distributed across different combinations of 'lpd_vote', 'gpd_vote', and 'grda_vote'. The chart helps explore relationships and proportions within the data, showcasing the contribution of each subcategory to the overall distribution of patients.","metadata":{"_uuid":"f53d6779-0308-4f8f-8b27-a0e8eef8edbd","_cell_guid":"305e63eb-be7d-4ea1-85bc-6ebe0312fbd6","trusted":true}},{"cell_type":"code","source":"fig = px.strip(df, x=\"patient_id\", y=\"seizure_vote\")\nfig.show()","metadata":{"_uuid":"557ed081-57ac-4450-bd88-020605fd530e","_cell_guid":"93c6ac0e-5d37-48b6-bca9-5d98b218ea74","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"px.scatter(\ndata_frame=df,\nx='spectrogram_label_offset_seconds',\ny='eeg_label_offset_seconds', \nanimation_frame='patient_id', \nrange_x=[30,90], \nrange_y=[-2000,60000]\n)","metadata":{"_uuid":"b7fa8e8a-2e6b-49d0-96d1-9e5be6b104a5","_cell_guid":"08c60221-d452-406e-8db0-c0e1029b27be","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"MODEL BUILDIND MODEL EVALUATION","metadata":{"_uuid":"31594705-aa48-4a92-8345-a01d45344bf4","_cell_guid":"10079729-b964-4ebd-86eb-ebc98e1e03d3","trusted":true}},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score, classification_report","metadata":{"_uuid":"2782d84d-222f-428f-9f0f-3a4a407a2e39","_cell_guid":"2ac974f1-7c29-4c16-99ae-594516c4fd91","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 5: Create a new column 'harmful' based on predictions\ndf['Harmful'] = df['other_vote']","metadata":{"_uuid":"b68fce50-fa07-422b-b4d4-1a9b1294316d","_cell_guid":"ca741fcb-d1cf-4ff5-ae25-628c32956563","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"_uuid":"da37e15b-ae88-4e78-9494-346fb3108f45","_cell_guid":"f14b6dbd-c56f-42d9-9ff0-25f4ffbfce30","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"_uuid":"70006b10-025e-4dec-ac23-9f835de347d3","_cell_guid":"fac6f0c5-41c0-4c40-ae50-785a39a15b3e","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.columns)","metadata":{"_uuid":"56936ecd-ebd0-4bdf-8e07-dc19e2001bf6","_cell_guid":"73d0b09e-7376-459d-bd6a-440297ee3229","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"_uuid":"ba9c25c1-3384-49af-8a4c-cea97f4aed36","_cell_guid":"9090853f-7138-4632-96c2-c94eab4d85e2","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display all column names in your DataFrame\nprint(df.columns)\n\n# Assuming 'expert_consensus' is in the column names, encode it\nif 'expert_consensus' in df.columns:\n    from sklearn.preprocessing import LabelEncoder\n\n    # Assuming 'expert_consensus' is a categorical column in your DataFrame\n    label_encoder = LabelEncoder()\n    df['expert_consensus_encoded'] = label_encoder.fit_transform(df['expert_consensus'])\n\n    # Drop the original 'expert_consensus' column\n    df = df.drop(['expert_consensus'], axis=1)\n    \n    # Display the modified DataFrame\n    print(df.head())\nelse:\n    print(\"Column 'expert_consensus' not found in the DataFrame.\")","metadata":{"_uuid":"df03f33a-9161-4d9b-9919-ebd1c1830d61","_cell_guid":"b23f1bb9-7ced-4b0b-8d8f-863683564d43","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = df.drop([\"Harmful\"], axis=1)","metadata":{"_uuid":"7c069b74-3c12-4e5d-910f-15e3fe5b4bd8","_cell_guid":"e0a9c9e9-c32e-423a-a42c-edef09a5ad9a","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = df[\"Harmful\"]","metadata":{"_uuid":"cc100e88-8e12-46c8-a716-52fa6783e904","_cell_guid":"6b57b12d-c8fb-4cd6-bc0a-a5cc5718cf8e","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x","metadata":{"_uuid":"b0769383-40c0-41f9-9aa3-8785da8a1a91","_cell_guid":"46fee297-5fc9-46d2-8a8d-6130bd9ad620","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y","metadata":{"_uuid":"61f47945-8170-4678-93da-e5704f4505f1","_cell_guid":"c486bc0b-3b70-45cc-b33b-fff88bcdeb5d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 2: Train-Test Split\nxtrain,xtest,ytrain,ytest = train_test_split(x , y , test_size=0.20, random_state=42)","metadata":{"_uuid":"a01eea9a-c535-48e5-a109-5744872dee04","_cell_guid":"5dfb24df-5e20-422a-93e1-40596df17223","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3: Initialize and Train the Model\nrfc = RandomForestClassifier(n_estimators=100,random_state=42)","metadata":{"_uuid":"d719e5d8-505c-4f69-8489-96878eeb1a3a","_cell_guid":"ea10d4fd-a8ef-47f2-af93-f042e964fa49","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc.fit(xtrain,ytrain)","metadata":{"_uuid":"769b58cb-1620-4515-bfb7-3dcc176f59f6","_cell_guid":"1b1c83a9-b4c5-4275-baf3-248b7c1ae391","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 4: Make Predictions\nypred = rfc.predict(xtest)\n\n# Step 5: Evaluate the Model\naccuracy = accuracy_score(ytest, ypred)\nprint(f\"Accuracy: {accuracy:.2f}\")\n\n# Additional evaluation metrics\nprint(\"Classification Report:\\n\", classification_report(ytest, ypred))","metadata":{"_uuid":"3a395a2c-5bad-4a31-859f-c8f74b717706","_cell_guid":"f342bf71-dcff-4ce4-8739-8b7ccedcc98c","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The accuracy of your model is reported as 1.00, which means the model is achieving perfect accuracy on the provided dataset. Here's a breakdown of the classification report metrics:\n\nPrecision: Precision is the ratio of correctly predicted positive observations to the total predicted positives. It indicates the accuracy of the positive predictions.\n\nRecall: Recall (Sensitivity or True Positive Rate) is the ratio of correctly predicted positive observations to all actual positives. It indicates the model's ability to capture all positive instances.\n\nF1-score: The F1-score is the harmonic mean of precision and recall. It provides a balance between precision and recall.\n\nSupport: The number of actual occurrences of each class in the specified dataset.\n\nHere's a summary of the classification report:\n\nThe model performs exceptionally well across all classes with high precision, recall, and F1-score values.\nThe weighted average and macro average metrics are also high, indicating overall strong performance.\nIt's worth noting that achieving perfect accuracy can sometimes indicate overfitting, especially if the model has not been tested on an independent dataset. Ensure that the model generalizes well to new, unseen data. If the dataset used for training is small, consider using techniques such as cross-validation and testing on a larger, diverse dataset to validate the model's performance.","metadata":{"_uuid":"52b6e4e8-e247-4076-8216-8a1ce6bd28c5","_cell_guid":"b7df6f7a-8931-4366-a93f-97f699696cc2","trusted":true}},{"cell_type":"code","source":"\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import classification_report","metadata":{"_uuid":"8cc2ccf7-19c0-4d11-a06c-07a47d812079","_cell_guid":"0659aa30-e66f-421a-bc02-ac75040909d5","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nknn = KNeighborsClassifier()\nsvm = SVC()\ndt=DecisionTreeClassifier()","metadata":{"_uuid":"582eed1b-7587-4ad3-bf56-f66cda2da0b3","_cell_guid":"46a8a020-dd7f-421e-b20d-4c05e6cd784c","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def mymodel(model):\n model.fit(xtrain, ytrain)\n ypred = model.predict(xtest)\n print(classification_report(ytest, ypred))\n return model","metadata":{"_uuid":"01875bc8-8454-4940-aaa3-79607da56b89","_cell_guid":"665dabe7-ffcb-4cb8-8719-1545b0262bfa","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mymodel(knn)","metadata":{"_uuid":"95a525fb-554f-4c92-97cb-c7cea427cdd1","_cell_guid":"cc28f3d5-b3d1-4e42-b96b-bbf7d10aa4ff","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mymodel(svm)","metadata":{"_uuid":"3bc5d7a3-c366-4325-a2f9-a112335dbfd5","_cell_guid":"3dfa5458-033f-4521-9f60-fb6118facf20","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mymodel(dt)","metadata":{"_uuid":"0b228508-880f-4d24-a5dd-b5b9735a6af6","_cell_guid":"c97b2149-84f3-47a6-9750-0cac864c4ffb","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeRegressor\nfrom sklearn import tree\n# Create a decision tree regressor\ndt_model = DecisionTreeRegressor()\n\n# Fit the model on the training data\ndt_model.fit(xtrain, ytrain)\n\n# Visualize the decision tree\nplt.figure(figsize=(15, 10))\ntree.plot_tree(dt_model, feature_names=xtrain.columns, filled=True, rounded=True)\nplt.show()","metadata":{"_uuid":"2be3bf32-7b98-4ba8-8e35-d45e7e7ce581","_cell_guid":"15b6b89d-0baf-4819-9001-e36e4451485b","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.cluster import KMeans\nfrom sklearn.preprocessing import StandardScaler\n\n# Assuming xtrain is your feature data\n# You may need to preprocess your data, for example, by scaling it\nscaler = StandardScaler()\nxtrain_scaled = scaler.fit_transform(xtrain)\n\n# Initialize a list to store the within-cluster sum of squares (WCSS) for different k values\nwcss = []\n\n# Try different values of k (number of clusters)\nfor k in range(1, 11):\n    kmeans = KMeans(n_clusters=k, random_state=42)\n    kmeans.fit(xtrain_scaled)\n    wcss.append(kmeans.inertia_)\n\n# Plot the elbow graph\nplt.figure(figsize=(8, 6))\nplt.plot(range(1, 11), wcss, marker='o')\nplt.title('Elbow Method for Optimal k')\nplt.xlabel('Number of Clusters (k)')\nplt.ylabel('Within-Cluster Sum of Squares (WCSS)')\nplt.show()","metadata":{"_uuid":"f07408b7-9ad0-49b5-bd2f-9092a92e2d16","_cell_guid":"549413db-021c-46c5-ad04-5bd167e98d57","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\n# Create an output directory if it doesn't exist\noutput_directory = 'kaggle kernels output purushv/hms-harmful-brain-activity-classification -p /path/to/dest/kaggle/working/output/'\nos.makedirs(output_directory, exist_ok=True)\n\n# Save a DataFrame to a CSV file in the output directory\ndf.to_csv(os.path.join(output_directory, 'output_file.csv'), index=False)\n","metadata":{"_uuid":"079dd5ff-d001-41a1-87ad-d3d96f39ff90","_cell_guid":"6790af31-66cd-4831-82bf-d88f3f80e34d","collapsed":false,"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}