{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# data:\n* In order to using sensor data, I make PCA embedding. But because data is too big, we use only 1 batch to make data.\n* Before embedding, **I create a matrix with `events` as rows, `sensor_id` as columns and `charge` as value of cells**. Strong charge values mean sensor signal is very strong, and its data will be accurate than others. Then I feed this matrix into sklearn PCA model\n* please upvote for this notebook if you use my embedding data\n\n# traning model and tuning:\n* please check this notebook: https://www.kaggle.com/astrung/pcaembed-gpu-catboost-optuna-approach\n* If you have another embedding approach, please suggest in my topic: https://www.kaggle.com/competitions/icecube-neutrinos-in-deep-ice/discussion/381078","metadata":{}},{"cell_type":"code","source":"%matplotlib inline\n\nimport os\nimport glob\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nPATH_DATASET = \"/kaggle/input/icecube-neutrinos-in-deep-ice\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-24T02:48:58.431090Z","iopub.execute_input":"2023-01-24T02:48:58.431654Z","iopub.status.idle":"2023-01-24T02:48:58.464633Z","shell.execute_reply.started":"2023-01-24T02:48:58.431545Z","shell.execute_reply":"2023-01-24T02:48:58.463678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta_train = pd.read_parquet(os.path.join(PATH_DATASET, \"train_meta.parquet\"))\nmeta_train_ = meta_train[meta_train['batch_id'] == 10]\nmeta_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-24T02:48:58.466818Z","iopub.execute_input":"2023-01-24T02:48:58.467536Z","iopub.status.idle":"2023-01-24T02:49:48.462340Z","shell.execute_reply.started":"2023-01-24T02:48:58.467488Z","shell.execute_reply":"2023-01-24T02:49:48.460552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### use only batch 10","metadata":{"_kg_hide-input":true}},{"cell_type":"code","source":"from tqdm.auto import tqdm\n\ndef transform_batch(path_batch_parquet, nb_sensors=5160):\n    df = pd.read_parquet(path_batch_parquet)\n    data = np.zeros((len(df.index.unique()), nb_sensors), dtype=np.float16)\n    event_ids = []\n    for i, (idx, dfg) in tqdm(enumerate(df.groupby(level=0))):\n        event_ids.append(idx)\n        data[i, dfg['sensor_id']] = dfg['charge']\n    return event_ids, data","metadata":{"execution":{"iopub.status.busy":"2023-01-24T02:49:49.186249Z","iopub.execute_input":"2023-01-24T02:49:49.186992Z","iopub.status.idle":"2023-01-24T02:49:49.308494Z","shell.execute_reply.started":"2023-01-24T02:49:49.186942Z","shell.execute_reply":"2023-01-24T02:49:49.307279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"event_ids, data = transform_batch(os.path.join(PATH_DATASET, \"train/batch_10.parquet\"))\nprint(f\"data size: {data.shape}\")\n\nmeta_train_ = meta_train[meta_train['event_id'].isin(event_ids)]\nmeta_train_ = dict(zip(meta_train_['event_id'].values, meta_train_[[\"azimuth\", \"zenith\"]].values.tolist()))\nangles = np.array([meta_train_[eid] for eid in event_ids], dtype=np.float16)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T02:49:49.309855Z","iopub.execute_input":"2023-01-24T02:49:49.310970Z","iopub.status.idle":"2023-01-24T02:50:44.862060Z","shell.execute_reply.started":"2023-01-24T02:49:49.310924Z","shell.execute_reply":"2023-01-24T02:50:44.860623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del meta_train, meta_train_\nimport gc\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data preprocessing","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(data, angles, train_size=0.9)\ndel data, angles","metadata":{"execution":{"iopub.status.busy":"2023-01-24T02:50:44.864041Z","iopub.execute_input":"2023-01-24T02:50:44.864427Z","iopub.status.idle":"2023-01-24T02:50:47.015738Z","shell.execute_reply.started":"2023-01-24T02:50:44.864395Z","shell.execute_reply":"2023-01-24T02:50:47.014317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\n\npreprocess = Pipeline([\n    ('scaler', StandardScaler()),\n    (\"PCA\", PCA(\n        n_components=510,\n        copy=False,\n    )),\n])\n\nX_train = preprocess.fit_transform(X_train)\nX_test = preprocess.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T02:50:47.017380Z","iopub.execute_input":"2023-01-24T02:50:47.018361Z","iopub.status.idle":"2023-01-24T02:56:29.643898Z","shell.execute_reply.started":"2023-01-24T02:50:47.018317Z","shell.execute_reply":"2023-01-24T02:56:29.642062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.save('/kaggle/working/x_train', X_train)\nnp.save('/kaggle/working/x_test', X_test)\nnp.save('/kaggle/working/y_train', y_train)\nnp.save('/kaggle/working/y_test', y_test)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T03:36:49.740996Z","iopub.execute_input":"2023-01-24T03:36:49.741572Z","iopub.status.idle":"2023-01-24T03:36:53.879479Z","shell.execute_reply.started":"2023-01-24T03:36:49.741526Z","shell.execute_reply":"2023-01-24T03:36:53.878073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\nwith open('/kaggle/working/preprocessor', 'wb') as outp: \n    pickle.dump(preprocess, outp, pickle.HIGHEST_PROTOCOL)","metadata":{"execution":{"iopub.status.busy":"2023-01-24T03:40:43.920039Z","iopub.execute_input":"2023-01-24T03:40:43.920560Z","iopub.status.idle":"2023-01-24T03:40:43.964326Z","shell.execute_reply.started":"2023-01-24T03:40:43.920522Z","shell.execute_reply":"2023-01-24T03:40:43.963330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}