{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":154204277,"sourceType":"kernelVersion"},{"sourceId":155730565,"sourceType":"kernelVersion"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport numpy as np\nfrom matplotlib.colors import LogNorm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-03T09:29:53.658959Z","iopub.execute_input":"2024-02-03T09:29:53.659363Z","iopub.status.idle":"2024-02-03T09:29:54.609373Z","shell.execute_reply.started":"2024-02-03T09:29:53.659331Z","shell.execute_reply":"2024-02-03T09:29:54.608308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")\ndf","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:54.611257Z","iopub.execute_input":"2024-02-03T09:29:54.611715Z","iopub.status.idle":"2024-02-03T09:29:54.871335Z","shell.execute_reply.started":"2024-02-03T09:29:54.611681Z","shell.execute_reply":"2024-02-03T09:29:54.870235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Distribution ","metadata":{}},{"cell_type":"code","source":"plt.hist(df['expert_consensus'])\nplt.title(\"Expert consensus distribution\")","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:54.872525Z","iopub.execute_input":"2024-02-03T09:29:54.872846Z","iopub.status.idle":"2024-02-03T09:29:55.221560Z","shell.execute_reply.started":"2024-02-03T09:29:54.872817Z","shell.execute_reply":"2024-02-03T09:29:55.220413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes = [\n    \"seizure_vote\",\n    \"lpd_vote\",\n    \"lrda_vote\",\n    \"gpd_vote\",\n    \"grda_vote\",\n    \"other_vote\"\n]","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:55.223908Z","iopub.execute_input":"2024-02-03T09:29:55.224336Z","iopub.status.idle":"2024-02-03T09:29:55.228367Z","shell.execute_reply.started":"2024-02-03T09:29:55.224307Z","shell.execute_reply":"2024-02-03T09:29:55.227543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[classes].max()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:55.229386Z","iopub.execute_input":"2024-02-03T09:29:55.229681Z","iopub.status.idle":"2024-02-03T09:29:55.247493Z","shell.execute_reply.started":"2024-02-03T09:29:55.229655Z","shell.execute_reply":"2024-02-03T09:29:55.246770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[classes].mean()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:55.248836Z","iopub.execute_input":"2024-02-03T09:29:55.249252Z","iopub.status.idle":"2024-02-03T09:29:55.261428Z","shell.execute_reply.started":"2024-02-03T09:29:55.249214Z","shell.execute_reply":"2024-02-03T09:29:55.260354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(df[classes], label=classes)\nplt.title(\"distribution of votes\")\nplt.legend()\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:55.262588Z","iopub.execute_input":"2024-02-03T09:29:55.262921Z","iopub.status.idle":"2024-02-03T09:29:55.737552Z","shell.execute_reply.started":"2024-02-03T09:29:55.262893Z","shell.execute_reply":"2024-02-03T09:29:55.736553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### The distribution of votes is not uniform...\nThis is important, the number of votes determine the target distribution so you might have to do some balanced sampling when training.. or the good old data augmentation, i'm pretty sure we can add noise to the same label or something like that... ","metadata":{"execution":{"iopub.status.busy":"2024-01-31T17:55:44.550963Z","iopub.execute_input":"2024-01-31T17:55:44.552387Z","iopub.status.idle":"2024-01-31T17:55:44.572051Z","shell.execute_reply.started":"2024-01-31T17:55:44.552335Z","shell.execute_reply":"2024-01-31T17:55:44.570717Z"}}},{"cell_type":"code","source":"def n_votes_in_range(upper):\n    n_df = df.copy()\n    for vote_class in classes:\n\n        n_df = n_df[n_df[vote_class]<upper]\n    return len(n_df)/len(df)","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:55.738856Z","iopub.execute_input":"2024-02-03T09:29:55.739194Z","iopub.status.idle":"2024-02-03T09:29:55.744587Z","shell.execute_reply.started":"2024-02-03T09:29:55.739163Z","shell.execute_reply":"2024-02-03T09:29:55.743469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percentage_votes_in_ranges_cumulative = []\nfor i in range(1,25):\n    percentage_votes_in_ranges_cumulative.append(n_votes_in_range(i))    ","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:55.745988Z","iopub.execute_input":"2024-02-03T09:29:55.746409Z","iopub.status.idle":"2024-02-03T09:29:56.513920Z","shell.execute_reply.started":"2024-02-03T09:29:55.746380Z","shell.execute_reply":"2024-02-03T09:29:56.512818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"percentage_votes_in_ranges_cumulative","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:56.517942Z","iopub.execute_input":"2024-02-03T09:29:56.518383Z","iopub.status.idle":"2024-02-03T09:29:56.524608Z","shell.execute_reply.started":"2024-02-03T09:29:56.518341Z","shell.execute_reply":"2024-02-03T09:29:56.523614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(percentage_votes_in_ranges_cumulative, label=\"percentage_votes_in_ranges_cumulative\")\nplt.title(\"no of votes range X percentage of data coverged\")","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:56.526119Z","iopub.execute_input":"2024-02-03T09:29:56.526455Z","iopub.status.idle":"2024-02-03T09:29:56.799823Z","shell.execute_reply.started":"2024-02-03T09:29:56.526418Z","shell.execute_reply":"2024-02-03T09:29:56.798732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KL Divergence\n\n### $D_{KL}(P_{true} || Q_{pred}) = \\sum_{i}^{n\\_classes}{P(i)\\cdot log(\\frac{P(i)}{Q(i)})}$","metadata":{}},{"cell_type":"code","source":"def KL_divergence(train_probs, predicted_probs):\n#   expecting a numpy array\n# im unsure on the mean, i saw the official do it.. but if you think its wrong, please lmk\n    return (train_probs*np.log(train_probs/predicted_probs)).sum(axis=1).mean()\n    ","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:56.801622Z","iopub.execute_input":"2024-02-03T09:29:56.802437Z","iopub.status.idle":"2024-02-03T09:29:56.808381Z","shell.execute_reply.started":"2024-02-03T09:29:56.802392Z","shell.execute_reply":"2024-02-03T09:29:56.807179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"true_probs =df[classes].values\neps=1e-9\ntrue_probs = (true_probs/true_probs.sum(axis=1)[:,np.newaxis])+eps\nuniform_probs = np.ones_like(true_probs)/6\n\nprint(\"Dkl(TrueProbs||UniformPreds)\",KL_divergence(true_probs, uniform_probs))\nprint(\"Dkl(TrueProbs||TrueProbs) should be zero\",KL_divergence(true_probs, true_probs))","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:56.809827Z","iopub.execute_input":"2024-02-03T09:29:56.810172Z","iopub.status.idle":"2024-02-03T09:29:56.839719Z","shell.execute_reply.started":"2024-02-03T09:29:56.810142Z","shell.execute_reply":"2024-02-03T09:29:56.838549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Vote Entropy","metadata":{}},{"cell_type":"code","source":"def get_entropy(df):\n    true_probs = df[classes].values\n    eps=1e-9\n    true_probs_corrected = (true_probs/true_probs.sum(axis=1)[:,np.newaxis])+eps # corrected becuz i dont want np.log(0)    \n    return abs(-(np.log(true_probs_corrected)*true_probs).sum(axis=1)) # absolute because i added eps... -log(1) = 0, but -log(1.0001) is 0.00001 or smth\n\n\n    ","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:56.841065Z","iopub.execute_input":"2024-02-03T09:29:56.841931Z","iopub.status.idle":"2024-02-03T09:29:56.847172Z","shell.execute_reply.started":"2024-02-03T09:29:56.841896Z","shell.execute_reply":"2024-02-03T09:29:56.846075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(get_entropy(df), bins=25)\nplt.title(\"No. of votes X Entropy (all data)\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:56.848888Z","iopub.execute_input":"2024-02-03T09:29:56.849463Z","iopub.status.idle":"2024-02-03T09:29:57.132731Z","shell.execute_reply.started":"2024-02-03T09:29:56.849393Z","shell.execute_reply":"2024-02-03T09:29:57.131704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### so most votes are entropy 0 (meaning all votes being given to a single class)\nWe must also see how the ones with entropy 0 are distributed (that's the cleanest data right...)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T06:39:31.297614Z","iopub.execute_input":"2024-02-03T06:39:31.297882Z","iopub.status.idle":"2024-02-03T06:39:31.301032Z","shell.execute_reply.started":"2024-02-03T06:39:31.297859Z","shell.execute_reply":"2024-02-03T06:39:31.300464Z"}}},{"cell_type":"code","source":"all_entropies = get_entropy(df)\nn_votes_zero_entropy= df[all_entropies<1e-5][classes].sum(axis=1)\n\n\n\nplt.hist(n_votes_zero_entropy, bins=25)\nplt.xticks(list(range(0,26)))\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:57.134280Z","iopub.execute_input":"2024-02-03T09:29:57.134619Z","iopub.status.idle":"2024-02-03T09:29:57.489498Z","shell.execute_reply.started":"2024-02-03T09:29:57.134566Z","shell.execute_reply":"2024-02-03T09:29:57.488389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## As we dive deeper\nYou can see above that even though most votes were entropy zero, the number of votes mostly given were only 3 ","metadata":{}},{"cell_type":"markdown","source":"# full consensus with 10 or more votes \nIt seems like most votes are 3... which really really sucks ngl","metadata":{}},{"cell_type":"code","source":"print(\"all votes with complete expert consensus:\", len(n_votes_zero_entropy))\nprint(\"all votes with complete strong consensus(10+ votes):\", len(n_votes_zero_entropy[n_votes_zero_entropy>=10]))\n\n","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:57.493355Z","iopub.execute_input":"2024-02-03T09:29:57.493830Z","iopub.status.idle":"2024-02-03T09:29:57.500541Z","shell.execute_reply.started":"2024-02-03T09:29:57.493787Z","shell.execute_reply":"2024-02-03T09:29:57.499482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(df[all_entropies<1e-5][n_votes_zero_entropy>=10]['expert_consensus'])","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:29:57.502164Z","iopub.execute_input":"2024-02-03T09:29:57.502835Z","iopub.status.idle":"2024-02-03T09:29:57.750291Z","shell.execute_reply.started":"2024-02-03T09:29:57.502795Z","shell.execute_reply":"2024-02-03T09:29:57.749339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### What you see above is even more weird\nWe don't have any real STRONG consensus on seizures, GRDAs and LRDAs.. (i thought going in the distribution would be uniform but we don't live in a perfect world LOL)","metadata":{}},{"cell_type":"markdown","source":"# No of votes / entropy (heatmap) How no. of votes change as the consensus becomes disagreeable","metadata":{}},{"cell_type":"code","source":"step = 1\nranges=[(0,1e-5)]\nranges+=[(i-1, i) for i in range(2,31, step)]\ntotal_votes =len(df)\n# range X discharge_type X n_votes\nvotes_trend_heatmap = {}\nfor (l, h) in ranges:\n    all_entropies = get_entropy(df)\n\n    vote_groups = df[(all_entropies>l)  & (all_entropies<h)][classes].sum(axis=0)\n    \n    votes_trend_heatmap[(l, h)]  = vote_groups\nvotes_trend_heatmap = pd.DataFrame(votes_trend_heatmap)\n","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:57.895683Z","iopub.execute_input":"2024-02-03T09:29:57.896106Z","iopub.status.idle":"2024-02-03T09:29:58.318903Z","shell.execute_reply.started":"2024-02-03T09:29:57.896071Z","shell.execute_reply":"2024-02-03T09:29:58.317906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,3))\nplt.imshow(votes_trend_heatmap.values,cmap='plasma', aspect='auto')\nplt.title(\"No of votes X Entropy\")\nplt.yticks(list(range(0,6)),classes)\nplt.ylabel(\"discharge type\")\nplt.xlabel(\"ENTROPY\")\n\n\nplt.colorbar()  \nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:58.655207Z","iopub.execute_input":"2024-02-03T09:29:58.655881Z","iopub.status.idle":"2024-02-03T09:29:58.976428Z","shell.execute_reply.started":"2024-02-03T09:29:58.655843Z","shell.execute_reply":"2024-02-03T09:29:58.975409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(12,3))\nplt.imshow(votes_trend_heatmap.values,cmap='plasma', aspect='auto',norm=LogNorm())\nplt.title(\"No of votes X Entropy [LogNormed]\")\nplt.yticks(list(range(0,6)),classes)\nplt.ylabel(\"discharge type\")\nplt.xlabel(\"ENTROPY\")\n\n\nplt.colorbar()  \nplt.show()","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2024-02-03T09:29:59.264464Z","iopub.execute_input":"2024-02-03T09:29:59.265477Z","iopub.status.idle":"2024-02-03T09:29:59.963471Z","shell.execute_reply.started":"2024-02-03T09:29:59.265424Z","shell.execute_reply":"2024-02-03T09:29:59.962430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### What I see across the board\nEven with very mixed consensus (Lrda_Grda) votes are in less quantity (this means that rhytmic discharges are less common so we have less data) ","metadata":{}},{"cell_type":"markdown","source":"# P - values\nNow let's determine the p-values of the votes, determining how convincing the votes are.\n(took some ideas from @dan3dewey), exploring the idea further...\n!!!! I didn't use 'good' ways to get probabilities from counts (only vote/total_votes) for each row.. it could be better...? suggest","metadata":{}},{"cell_type":"code","source":"vote_probs = df[classes].values/df[classes].sum(axis=1).values[:,np.newaxis]","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:30:42.508388Z","iopub.execute_input":"2024-02-03T09:30:42.508802Z","iopub.status.idle":"2024-02-03T09:30:42.536632Z","shell.execute_reply.started":"2024-02-03T09:30:42.508767Z","shell.execute_reply":"2024-02-03T09:30:42.535578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p_max = vote_probs.max(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:31:53.622069Z","iopub.execute_input":"2024-02-03T09:31:53.622458Z","iopub.status.idle":"2024-02-03T09:31:53.627683Z","shell.execute_reply.started":"2024-02-03T09:31:53.622428Z","shell.execute_reply":"2024-02-03T09:31:53.626877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"votes_x_threshold = {}\nfor n in np.linspace(0,1,10):\n    votes_x_threshold[n] = len(p_max[p_max>=n])\n    ","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:38:36.710702Z","iopub.execute_input":"2024-02-03T09:38:36.711151Z","iopub.status.idle":"2024-02-03T09:38:36.721159Z","shell.execute_reply.started":"2024-02-03T09:38:36.711117Z","shell.execute_reply":"2024-02-03T09:38:36.720068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(list(votes_x_threshold.keys()),votes_x_threshold.values())\nplt.xlabel(\"p-threshold\")\nplt.ylabel(\"no.of rows\")\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:38:37.137218Z","iopub.execute_input":"2024-02-03T09:38:37.137671Z","iopub.status.idle":"2024-02-03T09:38:37.373509Z","shell.execute_reply.started":"2024-02-03T09:38:37.137612Z","shell.execute_reply":"2024-02-03T09:38:37.372091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"votes_x_threshold","metadata":{"execution":{"iopub.status.busy":"2024-02-03T09:37:44.931190Z","iopub.execute_input":"2024-02-03T09:37:44.932415Z","iopub.status.idle":"2024-02-03T09:37:44.940963Z","shell.execute_reply.started":"2024-02-03T09:37:44.932356Z","shell.execute_reply":"2024-02-03T09:37:44.939622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Conclusion\nWhen filtering data for training, remember that you can have 'complete consensus' (meaning entropy = 0, probability = 1.0) but as we in the data, most votes are in the range of 3, so its not a strong consensus, you can filter further by the number of votes you consider to be 'strong', so only getting rows with votes greater than 10 and votes only given to a single option, that would be 'complete (fully agreed on) and strong (enough votes) consensus)\n","metadata":{}}]}