{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":59093,"databundleVersionId":7469972,"sourceType":"competition"}],"dockerImageVersionId":30635,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Generate training data for reference only. . .\n# 生成训练数据，仅供参考。。。","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-01-12T13:47:03.847047Z","iopub.execute_input":"2024-01-12T13:47:03.847984Z","iopub.status.idle":"2024-01-12T13:47:04.314300Z","shell.execute_reply.started":"2024-01-12T13:47:03.847941Z","shell.execute_reply":"2024-01-12T13:47:04.313201Z"}}},{"cell_type":"code","source":"import os \nimport pathlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport librosa\nfrom tqdm import tqdm","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = pathlib.Path(\"/kaggle/input/hms-harmful-brain-activity-classification\")\nos.listdir(base_dir)","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:06.477256Z","iopub.execute_input":"2024-01-12T13:47:06.478056Z","iopub.status.idle":"2024-01-12T13:47:06.487273Z","shell.execute_reply.started":"2024-01-12T13:47:06.478012Z","shell.execute_reply":"2024-01-12T13:47:06.486281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_eegs = \"/kaggle/input/hms-harmful-brain-activity-classification/train_eegs/\"\ntrain_spectrograms = \"/kaggle/input/hms-harmful-brain-activity-classification/train_spectrograms/\"\npath_train = \"/kaggle/input/hms-harmful-brain-activity-classification/train.csv\"\noutput_size = 500","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:09.441817Z","iopub.execute_input":"2024-01-12T13:47:09.442180Z","iopub.status.idle":"2024-01-12T13:47:09.447309Z","shell.execute_reply.started":"2024-01-12T13:47:09.442146Z","shell.execute_reply":"2024-01-12T13:47:09.446242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train = pd.read_csv(path_train)\nds_train.head(100)","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:11.922170Z","iopub.execute_input":"2024-01-12T13:47:11.922602Z","iopub.status.idle":"2024-01-12T13:47:12.248877Z","shell.execute_reply.started":"2024-01-12T13:47:11.922564Z","shell.execute_reply":"2024-01-12T13:47:12.247628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cols =['seizure_vote', 'lpd_vote','gpd_vote', 'lrda_vote', 'grda_vote', 'other_vote']","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:15.472088Z","iopub.execute_input":"2024-01-12T13:47:15.472520Z","iopub.status.idle":"2024-01-12T13:47:15.477976Z","shell.execute_reply.started":"2024-01-12T13:47:15.472484Z","shell.execute_reply":"2024-01-12T13:47:15.476819Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_ds = ds_train.groupby([\"eeg_id\",\"spectrogram_id\"])[target_cols].sum().reset_index()\ntemp_ds[\"all_vote\"] = temp_ds[target_cols].sum(1)\nfor col in target_cols:\n    temp_ds[col] = temp_ds[col]/temp_ds[\"all_vote\"]\ntemp_ds = temp_ds.drop(columns = [\"all_vote\"])\ntemp_ds[temp_ds[\"eeg_id\"] == 893864755]","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:18.161829Z","iopub.execute_input":"2024-01-12T13:47:18.162219Z","iopub.status.idle":"2024-01-12T13:47:18.228639Z","shell.execute_reply.started":"2024-01-12T13:47:18.162187Z","shell.execute_reply":"2024-01-12T13:47:18.227462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport numpy\ndef resize_linear(image_matrix, new_height:int, new_width:int):\n    \"\"\"Perform a pure-numpy linear-resampled resize of an image.\"\"\"\n    output_image = numpy.zeros((new_height, new_width), dtype=image_matrix.dtype)\n    original_height, original_width = image_matrix.shape\n    inv_scale_factor_y = original_height/new_height\n    inv_scale_factor_x = original_width/new_width\n\n    # This is an ugly serial operation.\n    for new_y in range(new_height):\n        for new_x in range(new_width):\n            # If you had a color image, you could repeat this with all channels here.\n            # Find sub-pixels data:\n            old_x = new_x * inv_scale_factor_x\n            old_y = new_y * inv_scale_factor_y\n            x_fraction = old_x - math.floor(old_x)\n            y_fraction = old_y - math.floor(old_y)\n\n            # Sample four neighboring pixels:\n            left_upper = image_matrix[math.floor(old_y), math.floor(old_x)]\n            right_upper = image_matrix[math.floor(old_y), min(image_matrix.shape[1] - 1, math.ceil(old_x))]\n            left_lower = image_matrix[min(image_matrix.shape[0] - 1, math.ceil(old_y)), math.floor(old_x)]\n            right_lower = image_matrix[min(image_matrix.shape[0] - 1, math.ceil(old_y)), min(image_matrix.shape[1] - 1, math.ceil(old_x))]\n\n            # Interpolate horizontally:\n            blend_top = (right_upper * x_fraction) + (left_upper * (1.0 - x_fraction))\n            blend_bottom = (right_lower * x_fraction) + (left_lower * (1.0 - x_fraction))\n            # Interpolate vertically:\n            final_blend = (blend_top * y_fraction) + (blend_bottom * (1.0 - y_fraction))\n            output_image[new_y, new_x] = final_blend\n    return output_image","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:23.853370Z","iopub.execute_input":"2024-01-12T13:47:23.853892Z","iopub.status.idle":"2024-01-12T13:47:23.871933Z","shell.execute_reply.started":"2024-01-12T13:47:23.853844Z","shell.execute_reply":"2024-01-12T13:47:23.870717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"list_train = []\nfor i in tqdm(range(temp_ds.shape[0])):\n    eeg_id = temp_ds.loc[i,\"eeg_id\"]\n    values_eeg = pd.read_parquet(train_eegs+str(eeg_id)+\".parquet\").values[:2000,:]\n    spectrogram_id = temp_ds.loc[i,\"spectrogram_id\"]\n    values_spc = pd.read_parquet(train_spectrograms+str(spectrogram_id)+\".parquet\").values[:4000,1:]\n    values_spc=np.clip(values_spc,np.exp(-6),np.exp(10))#最大值为89209464.0\n    values_spc= np.log(values_spc)#对数变换\n    v1 = resize_linear(values_eeg,output_size,20)\n    v2 = resize_linear(values_spc,output_size,100)\n    #print(np.concatenate((v1, v2), axis=1).shape)\n    list_train.append(np.concatenate((v1, v2), axis=1))","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:47:28.107841Z","iopub.execute_input":"2024-01-12T13:47:28.108264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"input_size = values.shape[0]\noutput_size = 500\nbin_size = input_size // output_size\nvalues.reshape((output_size, bin_size,values.shape[1])).mean(1)","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:06:52.938821Z","iopub.execute_input":"2024-01-12T13:06:52.939244Z","iopub.status.idle":"2024-01-12T13:06:52.949845Z","shell.execute_reply.started":"2024-01-12T13:06:52.939209Z","shell.execute_reply":"2024-01-12T13:06:52.948323Z"}}},{"cell_type":"code","source":"train_ds = np.array(list_train)","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:28:21.243558Z","iopub.execute_input":"2024-01-12T13:28:21.243945Z","iopub.status.idle":"2024-01-12T13:28:21.249938Z","shell.execute_reply.started":"2024-01-12T13:28:21.243917Z","shell.execute_reply":"2024-01-12T13:28:21.248769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:30:28.578457Z","iopub.execute_input":"2024-01-12T13:30:28.578826Z","iopub.status.idle":"2024-01-12T13:30:28.639308Z","shell.execute_reply.started":"2024-01-12T13:30:28.578801Z","shell.execute_reply":"2024-01-12T13:30:28.638322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(temp_ds[target_cols].values,\"target.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:32:09.128696Z","iopub.execute_input":"2024-01-12T13:32:09.129095Z","iopub.status.idle":"2024-01-12T13:32:09.145833Z","shell.execute_reply.started":"2024-01-12T13:32:09.129063Z","shell.execute_reply":"2024-01-12T13:32:09.144616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"joblib.dump(np.nan_to_num(train_ds),\"train_matrix.pkl\")","metadata":{"execution":{"iopub.status.busy":"2024-01-12T13:33:02.999000Z","iopub.execute_input":"2024-01-12T13:33:02.999412Z","iopub.status.idle":"2024-01-12T13:33:03.014214Z","shell.execute_reply.started":"2024-01-12T13:33:02.999381Z","shell.execute_reply":"2024-01-12T13:33:03.011811Z"},"trusted":true},"execution_count":null,"outputs":[]}]}