{"metadata":{"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"},{"sourceId":7526248,"sourceType":"datasetVersion","datasetId":4308295},{"sourceId":7718165,"sourceType":"datasetVersion","datasetId":4507864},{"sourceId":6127,"sourceType":"modelInstanceVersion","modelInstanceId":4598}],"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"papermill":{"default_parameters":{},"duration":3846.080383,"end_time":"2024-01-14T04:20:19.064569","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-01-14T03:16:12.984186","version":"2.4.0"},"widgets":{"application/vnd.jupyter.widget-state+json":{"state":{"08983a9c6aff42578980f4f7113c3ee2":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_4411aefc021d46d0ada7b645eb53ec48","placeholder":"​","style":"IPY_MODEL_09a10a8cf9334c51857397ed50398c8e","value":"Searching best thr : 100%"}},"09a10a8cf9334c51857397ed50398c8e":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"1f3989a0c01248328e16875075e9d1c4":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HBoxModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HBoxModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HBoxView","box_style":"","children":["IPY_MODEL_08983a9c6aff42578980f4f7113c3ee2","IPY_MODEL_22cfcc0a7cc6455fbf3bb7c788c8a4e1","IPY_MODEL_c8392e8075224e3b8a020a16c1a08447"],"layout":"IPY_MODEL_6cec9a2c2fac450d87248aed8dd62f86"}},"22cfcc0a7cc6455fbf3bb7c788c8a4e1":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"FloatProgressModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"FloatProgressModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"ProgressView","bar_style":"success","description":"","description_tooltip":null,"layout":"IPY_MODEL_dffe80502d954bdea0bbb6353dbf5515","max":20,"min":0,"orientation":"horizontal","style":"IPY_MODEL_7ce1b34a4f864a42a6619eec82311eb0","value":20}},"4411aefc021d46d0ada7b645eb53ec48":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"6cec9a2c2fac450d87248aed8dd62f86":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"7ce1b34a4f864a42a6619eec82311eb0":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"ProgressStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"ProgressStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","bar_color":null,"description_width":""}},"83fe40a0b8f047cc8602206909d42361":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}},"9384babdb7054d55aecdf3e989ddc926":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"DescriptionStyleModel","state":{"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"DescriptionStyleModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"StyleView","description_width":""}},"c8392e8075224e3b8a020a16c1a08447":{"model_module":"@jupyter-widgets/controls","model_module_version":"1.5.0","model_name":"HTMLModel","state":{"_dom_classes":[],"_model_module":"@jupyter-widgets/controls","_model_module_version":"1.5.0","_model_name":"HTMLModel","_view_count":null,"_view_module":"@jupyter-widgets/controls","_view_module_version":"1.5.0","_view_name":"HTMLView","description":"","description_tooltip":null,"layout":"IPY_MODEL_83fe40a0b8f047cc8602206909d42361","placeholder":"​","style":"IPY_MODEL_9384babdb7054d55aecdf3e989ddc926","value":" 20/20 [04:34&lt;00:00, 12.66s/it]"}},"dffe80502d954bdea0bbb6353dbf5515":{"model_module":"@jupyter-widgets/base","model_module_version":"1.2.0","model_name":"LayoutModel","state":{"_model_module":"@jupyter-widgets/base","_model_module_version":"1.2.0","_model_name":"LayoutModel","_view_count":null,"_view_module":"@jupyter-widgets/base","_view_module_version":"1.2.0","_view_name":"LayoutView","align_content":null,"align_items":null,"align_self":null,"border":null,"bottom":null,"display":null,"flex":null,"flex_flow":null,"grid_area":null,"grid_auto_columns":null,"grid_auto_flow":null,"grid_auto_rows":null,"grid_column":null,"grid_gap":null,"grid_row":null,"grid_template_areas":null,"grid_template_columns":null,"grid_template_rows":null,"height":null,"justify_content":null,"justify_items":null,"left":null,"margin":null,"max_height":null,"max_width":null,"min_height":null,"min_width":null,"object_fit":null,"object_position":null,"order":null,"overflow":null,"overflow_x":null,"overflow_y":null,"padding":null,"right":null,"top":null,"visibility":null,"width":null}}},"version_major":2,"version_minor":0}}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<center><img src=\"https://keras.io/img/logo-small.png\" alt=\"Keras logo\" width=\"100\"><br/>\nThis starter notebook is provided by the Keras team.</center>","metadata":{"execution":{"iopub.execute_input":"2024-01-10T05:24:31.308329Z","iopub.status.busy":"2024-01-10T05:24:31.307595Z","iopub.status.idle":"2024-01-10T05:24:31.313088Z","shell.execute_reply":"2024-01-10T05:24:31.312113Z","shell.execute_reply.started":"2024-01-10T05:24:31.308287Z"},"papermill":{"duration":0.011755,"end_time":"2024-01-14T03:16:16.447481","exception":false,"start_time":"2024-01-14T03:16:16.435726","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:30:32.363775Z","iopub.execute_input":"2024-03-06T06:30:32.364183Z","iopub.status.idle":"2024-03-06T06:30:32.368676Z","shell.execute_reply.started":"2024-03-06T06:30:32.364142Z","shell.execute_reply":"2024-03-06T06:30:32.367636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EEG = pd.read_csv('/kaggle/input/eegysd/EEG_y (1).csv')","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:30:32.665393Z","iopub.execute_input":"2024-03-06T06:30:32.666580Z","iopub.status.idle":"2024-03-06T06:30:32.759356Z","shell.execute_reply.started":"2024-03-06T06:30:32.666525Z","shell.execute_reply":"2024-03-06T06:30:32.758276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = EEG.drop_duplicates(subset='eeg_id', keep='first')","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:30:33.246512Z","iopub.execute_input":"2024-03-06T06:30:33.247144Z","iopub.status.idle":"2024-03-06T06:30:33.254611Z","shell.execute_reply.started":"2024-03-06T06:30:33.247115Z","shell.execute_reply":"2024-03-06T06:30:33.253799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:30:34.226193Z","iopub.execute_input":"2024-03-06T06:30:34.226558Z","iopub.status.idle":"2024-03-06T06:30:34.243282Z","shell.execute_reply.started":"2024-03-06T06:30:34.226530Z","shell.execute_reply":"2024-03-06T06:30:34.242311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.sample(frac=1).reset_index(drop=True)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:26.078540Z","iopub.execute_input":"2024-03-06T06:31:26.078910Z","iopub.status.idle":"2024-03-06T06:31:26.085461Z","shell.execute_reply.started":"2024-03-06T06:31:26.078883Z","shell.execute_reply":"2024-03-06T06:31:26.084630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\na = glob.glob('/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/*.parquet')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:26.844329Z","iopub.execute_input":"2024-03-06T06:31:26.844795Z","iopub.status.idle":"2024-03-06T06:31:26.909957Z","shell.execute_reply.started":"2024-03-06T06:31:26.844764Z","shell.execute_reply":"2024-03-06T06:31:26.909164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train loader\nimport pandas as pd\nimport os\n\n# Replace 'your_directory' with the actual directory path where your feature files are stored\ndata_directory = '/kaggle/input/hms-harmful-brain-activity-classification/train_eegs'\n\n# Specify the desired cutoff length\ncutoff_length = 10000\n\ndef load_features(eeg_id):\n    file_path = os.path.join(data_directory, f'{eeg_id}.parquet')\n\n    # Assuming your features are stored in CSV files, adjust accordingly if using a different format\n    if os.path.exists(file_path):\n        features_df = pd.read_parquet(file_path)\n\n        # Apply cutoff or padding to ensure a consistent length\n        if len(features_df) >= cutoff_length:\n            features_df = features_df[:cutoff_length]\n        else:\n            # If the length is less than the cutoff, pad with zeros or handle accordingly\n            features_df = pd.concat([features_df, pd.DataFrame(0, index=range(cutoff_length - len(features_df)), columns=features_df.columns)])\n\n        return features_df.values  # Return features as a numpy array\n    else:\n        print(f'File not found for eeg_id {eeg_id}')\n        return None  # Handle the case where the file is not found\n\n# Example usage:\neeg_id_example = 3911565283\nloaded_features_example = load_features(eeg_id_example)\nprint(f'Features for eeg_id {eeg_id_example}:\\n{loaded_features_example}')\nprint(f'Should be of shape: ({cutoff_length}, 20)')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:28.843285Z","iopub.execute_input":"2024-03-06T06:31:28.843682Z","iopub.status.idle":"2024-03-06T06:31:28.853020Z","shell.execute_reply.started":"2024-03-06T06:31:28.843651Z","shell.execute_reply":"2024-03-06T06:31:28.851946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\ntest","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:30.361698Z","iopub.execute_input":"2024-03-06T06:31:30.362078Z","iopub.status.idle":"2024-03-06T06:31:30.373433Z","shell.execute_reply.started":"2024-03-06T06:31:30.362047Z","shell.execute_reply":"2024-03-06T06:31:30.372517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten, Dense","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:31.399619Z","iopub.execute_input":"2024-03-06T06:31:31.400362Z","iopub.status.idle":"2024-03-06T06:31:31.405429Z","shell.execute_reply.started":"2024-03-06T06:31:31.400330Z","shell.execute_reply":"2024-03-06T06:31:31.404456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loaded_features_example.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:32.926905Z","iopub.execute_input":"2024-03-06T06:31:32.927256Z","iopub.status.idle":"2024-03-06T06:31:32.931490Z","shell.execute_reply.started":"2024-03-06T06:31:32.927223Z","shell.execute_reply":"2024-03-06T06:31:32.930540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import numpy as np\nchunk_size = 128\ntotal_rows = df.shape[0]\ncutoff_length = 10000\n\nfor i in range(0, total_rows, chunk_size):\n    # Select the chunk of data\n    chunk = df.iloc[i:i+chunk_size]\n\n    # Load features for the current chunk\n    X = np.stack([load_features(eeg_id) for eeg_id in chunk['eeg_id']], axis=0)\n\n    # Extract target variables (Y)\n    Y = chunk[['seizure_vote_normalized', 'lpd_vote_normalized', 'gpd_vote_normalized',\n               'lrda_vote_normalized', 'grda_vote_normalized', 'other_vote_normalized']]\n\n    # Split the data into training and testing sets\n    X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n\n    # Optionally, scale the features\n    scaler = StandardScaler()\n    X_train_scaled = X_train\n    X_test_scaled = X_test\n\n    print(\"X_train_scaled shape:\", X_train_scaled.shape)\n    print(\"X_test_scaled shape:\", X_test_scaled.shape)\n    \n    # Reshape the data for 1D convolutional layer\n    X_train_reshaped = X_train_scaled.reshape((X_train_scaled.shape[0], X_train_scaled.shape[1], X_train_scaled.shape[2]))\n    X_test_reshaped = X_test_scaled.reshape((X_test_scaled.shape[0], X_test_scaled.shape[1], X_test_scaled.shape[2]))\n\n    nan_rows = np.isnan(X_train_reshaped).any(axis=(1, 2))\n    X_train_reshaped = X_train_reshaped[~nan_rows]\n    Y_train = Y_train[~nan_rows]\n    if i>128:\n        print('breaked')\n        break","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:33.880168Z","iopub.execute_input":"2024-03-06T06:31:33.880521Z","iopub.status.idle":"2024-03-06T06:31:38.802167Z","shell.execute_reply.started":"2024-03-06T06:31:33.880494Z","shell.execute_reply":"2024-03-06T06:31:38.801166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Main code\n\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import mean_squared_error\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv1D, MaxPooling1D, Flatten, Dense\nimport keras\n# Assuming your data is stored in a DataFrame called df\n\n# Load features from the directory based on 'eeg_id' (replace 'your_directory' with the actual directory)\n\n# Initialize the model outside the loop\nmodel = Sequential()\nmodel.add(Conv1D(filters=128, kernel_size=10, activation='relu', input_shape=(X_train_reshaped.shape[1], X_train_reshaped.shape[2])))\nmodel.add(MaxPooling1D(pool_size=2))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dense(6, activation = 'softmax'))  # Number of output neurons corresponds to the number of target variables\nLOSS = keras.losses.KLDivergence()\n# Compile the model\nmodel.compile(optimizer='adam',loss=LOSS)\n\n# Iterate through chunks of 20 rows and continue training the CNN model\nchunk_size = 2000\ntotal_rows = df.shape[0]\n\n\nfor i in range(0, total_rows, chunk_size):\n    # Select the chunk of data\n    chunk = df.iloc[i:i+chunk_size]\n\n    # Load features for the current chunk\n    X = np.stack([load_features(eeg_id) for eeg_id in chunk['eeg_id']], axis=0)\n\n    # Extract target variables (Y)\n    Y = chunk[['seizure_vote_normalized', 'lpd_vote_normalized', 'gpd_vote_normalized',\n               'lrda_vote_normalized', 'grda_vote_normalized', 'other_vote_normalized']]\n\n    # Split the data into training and testing sets\n    X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n\n    # Optionally, scale the features\n    scaler = StandardScaler()\n    X_train_scaled = X_train\n    X_test_scaled = X_test\n\n    print(\"X_train_scaled shape:\", X_train_scaled.shape)\n    print(\"X_test_scaled shape:\", X_test_scaled.shape)\n    \n    # Reshape the data for 1D convolutional layer\n    X_train_reshaped = X_train_scaled.reshape((X_train_scaled.shape[0], X_train_scaled.shape[1], X_train_scaled.shape[2]))\n    X_test_reshaped = X_test_scaled.reshape((X_test_scaled.shape[0], X_test_scaled.shape[1], X_test_scaled.shape[2]))\n\n    nan_rows = np.isnan(X_train_reshaped).any(axis=(1, 2))\n    X_train_reshaped = X_train_reshaped[~nan_rows]\n    Y_train = Y_train[~nan_rows]\n    \n    # Train the model on the current chunk\n    model.fit(X_train_reshaped, Y_train, epochs=2, batch_size=16, validation_split=0.2)\n    \n    # Evaluate the model on the test set\n    predictions = model.predict(X_test_reshaped)\n    predictions[np.isnan(predictions)] = 0\n    mse = mean_squared_error(Y_test, predictions)\n    print(f'Mean Squared Error for chunk {i}-{i+chunk_size}: {mse}')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:31:38.803741Z","iopub.execute_input":"2024-03-06T06:31:38.804034Z","iopub.status.idle":"2024-03-06T06:38:12.656094Z","shell.execute_reply.started":"2024-03-06T06:31:38.804009Z","shell.execute_reply":"2024-03-06T06:38:12.654970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test loader\nimport pandas as pd\nimport os\n\n# Replace 'your_directory' with the actual directory path where your feature files are stored\ndata_directory = '/kaggle/input/hms-harmful-brain-activity-classification/test_eegs'\n\n# Specify the desired cutoff length\ncutoff_length = 10000\n\ndef load_feature(eeg_id):\n    file_path = os.path.join(data_directory, f'{eeg_id}.parquet')\n\n    # Assuming your features are stored in CSV files, adjust accordingly if using a different format\n    if os.path.exists(file_path):\n        features_df = pd.read_parquet(file_path)\n\n        # Apply cutoff or padding to ensure a consistent length\n        if len(features_df) >= cutoff_length:\n            features_df = features_df[:cutoff_length]\n        else:\n            # If the length is less than the cutoff, pad with zeros or handle accordingly\n            features_df = pd.concat([features_df, pd.DataFrame(0, index=range(cutoff_length - len(features_df)), columns=features_df.columns)])\n\n        return features_df.values  # Return features as a numpy array\n    else:\n        pass\n        return None  # Handle the case where the file is not found\n\n# Example usage:\neeg_id_example = 3911565283\nloaded_features_example = load_features(eeg_id_example)\nprint(f'Features for eeg_id {eeg_id_example}:\\n{loaded_features_example}')\nprint(f'Should be of shape: ({cutoff_length}, 20)')\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:10:59.131392Z","iopub.status.idle":"2024-03-06T06:10:59.131809Z","shell.execute_reply.started":"2024-03-06T06:10:59.131606Z","shell.execute_reply":"2024-03-06T06:10:59.131626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#finally works\n\n#for i in range(0, total_rows, chunk_size):\n    # Select the chunk of data\n    #chunk = df.iloc[i:i+chunk_size]\ndf = pd.read_csv('/kaggle/input/hms-harmful-brain-activity-classification/test.csv')\nall_predictions = []\nfor i in df.eeg_id:\n    print(i)\n    # Load features for the current chunk\n    X = np.stack([load_feature(i) for eeg_id in ['eeg_id']], axis=0)\n\n    # Extract target variables (Y)\n#Y = chunk[['seizure_vote_normalized', 'lpd_vote_normalized', 'gpd_vote_normalized',\n #              'lrda_vote_normalized', 'grda_vote_normalized', 'other_vote_normalized']]\n\n    # Split the data into training and testing sets\n    #X_train, X_test, Y_train, Y_test = train_test_split(X, Y, test_size=0.2, random_state=42)\n\n    # Optionally, scale the features\n#    scaler = StandardScaler()\n    X_train_scaled = X\n#    X_test_scaled = X_test\n\n #   print(\"X_train_scaled shape:\", X_train_scaled.shape)\n  #  print(\"X_test_scaled shape:\", X_test_scaled.shape)\n    \n    # Reshape the data for 1D convolutional layer\n    X_train_reshaped = X_train_scaled.reshape((X_train_scaled.shape[0], X_train_scaled.shape[1], X_train_scaled.shape[2]))\n#    X_test_reshaped = X_test_scaled.reshape((X_test_scaled.shape[0], X_test_scaled.shape[1], X_test_scaled.shape[2]))\n\n    nan_rows = np.isnan(X_train_reshaped).any(axis=(1, 2))\n    X_train_reshaped = X_train_reshaped[~nan_rows]\n    current_predictions = model.predict(X_train_reshaped)\n\n    # Append the predictions to the list\n    all_predictions.append(current_predictions)\n\n# Concatenate all predictions into a single NumPy array\nfinal_predictions = np.concatenate(all_predictions, axis=0)\n\n# Create a DataFrame with 'eeg_id' as an index and prediction columns\npredictions_df = pd.DataFrame(data=final_predictions, columns=['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote'],\n                               index=df['eeg_id'].values[:final_predictions.shape[0]])\n\n# Display the DataFrame\nprint(predictions_df)\n#    Y_train = Y_train[~nan_rows]\n    ","metadata":{"execution":{"iopub.status.busy":"2024-03-06T06:10:59.133861Z","iopub.status.idle":"2024-03-06T06:10:59.134234Z","shell.execute_reply.started":"2024-03-06T06:10:59.134052Z","shell.execute_reply":"2024-03-06T06:10:59.134068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df","metadata":{"execution":{"iopub.status.busy":"2024-03-06T05:42:17.217609Z","iopub.execute_input":"2024-03-06T05:42:17.218300Z","iopub.status.idle":"2024-03-06T05:42:17.230605Z","shell.execute_reply.started":"2024-03-06T05:42:17.218259Z","shell.execute_reply":"2024-03-06T05:42:17.229438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df = pd.DataFrame(data=final_predictions, columns=['seizure_vote', 'lpd_vote', 'gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote'],\n                               index=df['eeg_id'][:final_predictions.shape[0]]).reset_index(drop=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-06T05:42:18.637541Z","iopub.execute_input":"2024-03-06T05:42:18.637921Z","iopub.status.idle":"2024-03-06T05:42:18.644951Z","shell.execute_reply.started":"2024-03-06T05:42:18.637892Z","shell.execute_reply":"2024-03-06T05:42:18.643996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions_df = predictions_df.fillna(0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T05:42:20.584863Z","iopub.execute_input":"2024-03-06T05:42:20.585310Z","iopub.status.idle":"2024-03-06T05:42:20.589579Z","shell.execute_reply.started":"2024-03-06T05:42:20.585274Z","shell.execute_reply":"2024-03-06T05:42:20.588613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-03-06T05:42:21.536826Z","iopub.execute_input":"2024-03-06T05:42:21.537690Z","iopub.status.idle":"2024-03-06T05:42:21.545243Z","shell.execute_reply.started":"2024-03-06T05:42:21.537654Z","shell.execute_reply":"2024-03-06T05:42:21.544341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#predictions_df.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2024-03-06T05:43:29.987831Z","iopub.execute_input":"2024-03-06T05:43:29.988199Z","iopub.status.idle":"2024-03-06T05:43:29.996335Z","shell.execute_reply.started":"2024-03-06T05:43:29.988170Z","shell.execute_reply":"2024-03-06T05:43:29.995380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}