{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30646,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport pandas as pd\nimport numpy as np\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-15T20:38:07.706326Z","iopub.execute_input":"2024-02-15T20:38:07.707582Z","iopub.status.idle":"2024-02-15T20:38:08.702336Z","shell.execute_reply.started":"2024-02-15T20:38:07.707526Z","shell.execute_reply":"2024-02-15T20:38:08.701652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:38:08.704041Z","iopub.execute_input":"2024-02-15T20:38:08.704758Z","iopub.status.idle":"2024-02-15T20:38:08.990155Z","shell.execute_reply.started":"2024-02-15T20:38:08.704726Z","shell.execute_reply":"2024-02-15T20:38:08.989069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def average_votes_per_eeg(df: pd.DataFrame):\n    \"\"\"\n    Sum and average votes for each EEG\n    Creates new columns: averaged_XXX\n\n    \"\"\"\n\n    TARGETS = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\n\n    tmp = df.groupby(\"eeg_id\")[TARGETS].agg(\"sum\")\n    \n    # Create columns\n    for t in TARGETS:\n        df[\"averaged_\" + t] = -1\n\n    # Propagate\n    for i, row in tmp.iterrows():\n        for t in TARGETS:\n            df.loc[df.eeg_id == i, \"averaged_\" + t] = row[t]\n\n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:38:08.991534Z","iopub.execute_input":"2024-02-15T20:38:08.992606Z","iopub.status.idle":"2024-02-15T20:38:09.008366Z","shell.execute_reply.started":"2024-02-15T20:38:08.992580Z","shell.execute_reply":"2024-02-15T20:38:09.006340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = average_votes_per_eeg(df)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:38:09.015494Z","iopub.execute_input":"2024-02-15T20:38:09.016193Z","iopub.status.idle":"2024-02-15T20:38:44.898494Z","shell.execute_reply.started":"2024-02-15T20:38:09.016152Z","shell.execute_reply":"2024-02-15T20:38:44.897273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TARGETS = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\nPRED_TARGETS = [\"seizure_vote\", \"lpd_vote\", \"gpd_vote\", \"lrda_vote\", \"grda_vote\", \"other_vote\"]\nAVG_TARGETS = [\"averaged_\" + t for t in TARGETS]\n\ny_pred = df[PRED_TARGETS].values\ny_true = df[AVG_TARGETS].values\n\ny_pred_norm = y_pred / y_pred.sum(axis=1, keepdims=True)\ny_true_norm = y_true / y_true.sum(axis=1, keepdims=True)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:38:44.900021Z","iopub.execute_input":"2024-02-15T20:38:44.901492Z","iopub.status.idle":"2024-02-15T20:38:44.910819Z","shell.execute_reply.started":"2024-02-15T20:38:44.901441Z","shell.execute_reply":"2024-02-15T20:38:44.909888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def kl_divergence(solution, submission, epsilon: float):\n    # Clip both the min and max following Kaggle conventions for related metrics like log loss\n    # Clipping the max avoids cases where the loss would be infinite or undefined, clipping the min\n    # prevents users from playing games with the 20th decimal place of predictions.\n    submission = np.clip(submission, epsilon, 1 - epsilon)\n\n    y_nonzero_indices = solution != 0\n    solution = solution.astype(float, copy=True)\n    solution[y_nonzero_indices] = solution[y_nonzero_indices] * np.log(solution[y_nonzero_indices] / submission[y_nonzero_indices])\n    # Set the loss equal to zero where y_true equals zero following the scipy convention:\n    # https://docs.scipy.org/doc/scipy/reference/generated/scipy.special.rel_entr.html#scipy.special.rel_entr\n    solution[~y_nonzero_indices] = 0\n\n    return solution\n\nloss = kl_divergence(y_pred_norm, y_true_norm, 1e-15)\ndf[\"kl\"] = loss.sum(axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:39:09.263400Z","iopub.execute_input":"2024-02-15T20:39:09.263710Z","iopub.status.idle":"2024-02-15T20:39:09.289649Z","shell.execute_reply.started":"2024-02-15T20:39:09.263688Z","shell.execute_reply":"2024-02-15T20:39:09.288527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_by_loss = df.sort_values(\"kl\", ascending=False)\ndf_by_loss[[\"kl\", \"eeg_id\", \"eeg_label_offset_seconds\", *PRED_TARGETS, *AVG_TARGETS]].head(20)","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:40:12.387812Z","iopub.execute_input":"2024-02-15T20:40:12.388151Z","iopub.status.idle":"2024-02-15T20:40:12.434322Z","shell.execute_reply.started":"2024-02-15T20:40:12.388125Z","shell.execute_reply":"2024-02-15T20:40:12.433714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(figsize=(20, 8))\nplt.hist(df.kl, bins=100)\nplt.yscale(\"log\")","metadata":{"execution":{"iopub.status.busy":"2024-02-15T20:41:27.448282Z","iopub.execute_input":"2024-02-15T20:41:27.448603Z","iopub.status.idle":"2024-02-15T20:41:28.135860Z","shell.execute_reply.started":"2024-02-15T20:41:27.448577Z","shell.execute_reply":"2024-02-15T20:41:28.134848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}