{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":9594138,"sourceType":"datasetVersion","datasetId":5852146}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Import Libs","metadata":{}},{"cell_type":"code","source":"import os\nimport warnings # 避免一些可以忽略的报错\nwarnings.filterwarnings('ignore')\nimport random\nimport gc\nimport copy\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm # 进度条\nimport time\nfrom scipy import stats\n\nimport torch\nfrom torch import nn\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler # 学习率调度器\n\nimport timm # 预训练神经网络库，可直接调用预训练好的模型\nfrom PIL import Image\nimport albumentations as A # 数据增强库\nfrom albumentations.pytorch import ToTensorV2\n\nfrom collections import defaultdict # 记录 loss lr 等相关参数的变化\n# 改变 终端颜色 方便观察\nfrom colorama import Fore, Back, Style\nb_ = Fore.BLUE\nsr_ = Style.RESET_ALL","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:52:08.225231Z","iopub.execute_input":"2024-10-10T13:52:08.226112Z","iopub.status.idle":"2024-10-10T13:52:16.983984Z","shell.execute_reply.started":"2024-10-10T13:52:08.226045Z","shell.execute_reply":"2024-10-10T13:52:16.982820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CONFIG","metadata":{}},{"cell_type":"code","source":"is_debug = False\n\nclass CONFIG:\n    seed = 308\n    \n    epochs = 10 if not is_debug else 2\n    now_cv = 0\n    \n    train_batch_size = 2\n    valid_batch_size = 4\n    \n    use_length = 300000\n    per_row_use = 300\n    par_columns = ['X', 'Y', 'Z', 'enmo', 'anglez', 'non-wear_flag', 'light', 'battery_voltage', \n                   'time_of_day', 'weekday', 'quarter', 'relative_date_PCIAT']\n    \n    n_classes = 4\n\n    n_workers = os.cpu_count()\n    \n    learning_rate = 1e-3\n    weight_decay = 1e-6\n    scheduler = 'CosineAnnealingWithWarmupLR'\n    T_max = 33600 // train_batch_size * epochs \n    min_lr = 1e-6\n    \n    device = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\n    \n    train_csv = \"/kaggle/input/20241010-cmi-piu-mytrain-csv/CMIPIU_transformer_test.csv\"\n    par_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet\"\n    ckpt_save_path = \"/kaggle/working/output\"","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:00:11.649996Z","iopub.execute_input":"2024-10-10T14:00:11.650480Z","iopub.status.idle":"2024-10-10T14:00:11.658425Z","shell.execute_reply.started":"2024-10-10T14:00:11.650434Z","shell.execute_reply":"2024-10-10T14:00:11.657331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Set Random Seed","metadata":{}},{"cell_type":"code","source":"def set_seed(seed=308):\n    random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    \nset_seed(CONFIG.seed)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:52:16.995129Z","iopub.execute_input":"2024-10-10T13:52:16.995479Z","iopub.status.idle":"2024-10-10T13:52:17.011302Z","shell.execute_reply.started":"2024-10-10T13:52:16.995444Z","shell.execute_reply":"2024-10-10T13:52:17.010169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Progress","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv(CONFIG.train_csv)\ntrain","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:52:17.013624Z","iopub.execute_input":"2024-10-10T13:52:17.013969Z","iopub.status.idle":"2024-10-10T13:52:17.052776Z","shell.execute_reply.started":"2024-10-10T13:52:17.013933Z","shell.execute_reply":"2024-10-10T13:52:17.051729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset and DataLoader","metadata":{}},{"cell_type":"code","source":"class CMIPIUDataset(Dataset):\n    def __init__(self, df):\n        super().__init__()\n        self.df = df\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        row = self.df.iloc[idx, :]\n        _id = row.id\n        _label = row.label\n        \n        df_par = pd.read_parquet(os.path.join(CONFIG.par_path, 'id=' + _id))\n        df_par = df_par.fillna(0.0)\n        \n        if len(df_par) < CONFIG.use_length:\n            num_of_zeros = CONFIG.use_length - len(df_par)  # 补齐2行\n            zeros_df = pd.DataFrame(np.zeros((num_of_zeros, df_par.shape[1])), columns=df_par.columns)\n            df_par = pd.concat([df_par, zeros_df], ignore_index=True)\n        elif len(row) > CONFIG.use_length:\n            df_par = df_par.iloc[: CONFIG.use_length, :]\n            \n        total_data = []\n        data_len = CONFIG.use_length // CONFIG.per_row_use\n        for i in range(data_len):\n            start = i * CONFIG.per_row_use\n            end = (i + 1 ) * CONFIG.per_row_use\n            tmp = df_par.iloc[start: end, :]\n            tmp\n\n            per_data = []\n            for col in CONFIG.par_columns:\n                _mean = np.mean(tmp[col].values) # 平均值\n                _max = np.max(tmp[col].values) # 最大值\n                _min = np.min(tmp[col].values) # 最小值\n                _majority, count = stats.mode(tmp[col].values) # 众数\n                _var = np.var(tmp[col].values) # 方差\n                per_data += [_mean, _max, _min, _majority, _var]\n\n            total_data.append(np.array(per_data))\n        x = np.concatenate([total_data], axis=0) # (1000, 60)\n        y = _label\n        \n        x = torch.from_numpy(x)\n        y = torch.tensor(y)\n        \n        return x, y","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:52:17.054503Z","iopub.execute_input":"2024-10-10T13:52:17.055172Z","iopub.status.idle":"2024-10-10T13:52:17.067495Z","shell.execute_reply.started":"2024-10-10T13:52:17.055125Z","shell.execute_reply":"2024-10-10T13:52:17.066307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_loaders(df, fold=0):\n    df_train = df[df['kfold'] != fold]\n    df_valid = df[df['kfold'] == fold]\n    \n    train_datasets = CMIPIUDataset(df=df_train)\n    valid_datasets = CMIPIUDataset(df=df_valid)\n    \n    train_loader = DataLoader(train_datasets, batch_size=CONFIG.train_batch_size, num_workers=CONFIG.n_workers, shuffle=True, pin_memory=True)\n    valid_loader = DataLoader(valid_datasets, batch_size=CONFIG.valid_batch_size, num_workers=CONFIG.n_workers, shuffle=False, pin_memory=True)\n    \n    return train_loader, valid_loader","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:52:17.068734Z","iopub.execute_input":"2024-10-10T13:52:17.069062Z","iopub.status.idle":"2024-10-10T13:52:17.078480Z","shell.execute_reply.started":"2024-10-10T13:52:17.069029Z","shell.execute_reply":"2024-10-10T13:52:17.077463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 测试 dataset 和 dataloader 是否工作正常\nstart_time = time.time()\ntrain_loader, valid_loader = prepare_loaders(train)\nx_train, y_train = next(iter(train_loader))\nx_valid, y_valid = next(iter(valid_loader))\nprint(f\"X_train shape : {x_train.shape}\") # (batch_size, time, feature)\nprint(f\"y_train shape : {y_train.shape}\")\nprint(f\"x_valid shape : {x_valid.shape}\")\nprint(f\"y_valid shape : {y_valid.shape}\")\n\ndel train_loader, valid_loader, x_train, y_train, x_valid, y_valid\ngc.collect()\nend_time = time.time()\nused_time = end_time - start_time\nprint(f\"use {used_time // 60} min {used_time % 60: .2f} sec.\")","metadata":{"execution":{"iopub.status.busy":"2024-10-10T14:00:14.824597Z","iopub.execute_input":"2024-10-10T14:00:14.825359Z","iopub.status.idle":"2024-10-10T14:01:55.185425Z","shell.execute_reply.started":"2024-10-10T14:00:14.825317Z","shell.execute_reply":"2024-10-10T14:01:55.184093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if len(df) < CONFIG.use_length:\n    num_of_zeros = CONFIG.use_length - len(row)  # 补齐2行\n    zeros_df = pd.DataFrame(np.zeros((num_of_zeros, df.shape[1])), columns=df.columns)\n    df = pd.concat([df, zeros_df], ignore_index=True)\nelif len(row) > CONFIG.use_length:\n    df = df.iloc[: CONFIG.use_length, :]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:16:25.711858Z","iopub.execute_input":"2024-10-10T13:16:25.712317Z","iopub.status.idle":"2024-10-10T13:16:25.749879Z","shell.execute_reply.started":"2024-10-10T13:16:25.712273Z","shell.execute_reply":"2024-10-10T13:16:25.748526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:16:26.691979Z","iopub.execute_input":"2024-10-10T13:16:26.692433Z","iopub.status.idle":"2024-10-10T13:16:26.723004Z","shell.execute_reply.started":"2024-10-10T13:16:26.692390Z","shell.execute_reply":"2024-10-10T13:16:26.721824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = 'X'\nmean = np.var(tmp[col].values)\nmean\n\n# mode_value, count = stats.mode(tmp[col].values)\n# print(\"众数是:\", mode_value)\n# print(\"出现次数:\", count)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:25:47.108547Z","iopub.execute_input":"2024-10-10T13:25:47.108996Z","iopub.status.idle":"2024-10-10T13:25:47.117305Z","shell.execute_reply.started":"2024-10-10T13:25:47.108955Z","shell.execute_reply":"2024-10-10T13:25:47.116165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom scipy import stats\n\n# 示例数据\ndata = np.array([1, 2, 2, 3, 4, 4, 4, 5, 6])\n\n# 使用 scipy 计算众数\nmode_value, count = stats.mode(data)\n\nprint(\"众数是:\", mode_value)\nprint(\"出现次数:\", count)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:22:20.403379Z","iopub.execute_input":"2024-10-10T13:22:20.403921Z","iopub.status.idle":"2024-10-10T13:22:20.413832Z","shell.execute_reply.started":"2024-10-10T13:22:20.403875Z","shell.execute_reply":"2024-10-10T13:22:20.412494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 1\n\ndata_len = CONFIG.use_length // CONFIG.per_row_use\n\ntotal_data = []\nfor i in range(data_len):\n    start = i * CONFIG.per_row_use\n    end = (i + 1 ) * CONFIG.per_row_use\n    tmp = df.iloc[start: end, :]\n    tmp\n\n    per_data = []\n    for col in CONFIG.par_columns:\n        _mean = np.mean(tmp[col].values) # 平均值\n        _max = np.max(tmp[col].values) # 最大值\n        _min = np.min(tmp[col].values) # 最小值\n        _majority, count = stats.mode(data) # 众数\n        _var = np.var(tmp[col].values) # 方差\n        per_data += [_mean, _max, _min, _majority, _var]\n\n    total_data.append(np.array(per_data))\n    \nnp.concatenate(total_data, axis=1).shape/","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:33:19.966821Z","iopub.execute_input":"2024-10-10T13:33:19.967285Z","iopub.status.idle":"2024-10-10T13:33:29.536552Z","shell.execute_reply.started":"2024-10-10T13:33:19.967244Z","shell.execute_reply":"2024-10-10T13:33:29.535271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = np.concatenate([total_data], axis=0)\na.shape","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:34:47.597023Z","iopub.execute_input":"2024-10-10T13:34:47.598367Z","iopub.status.idle":"2024-10-10T13:34:47.607733Z","shell.execute_reply.started":"2024-10-10T13:34:47.598313Z","shell.execute_reply":"2024-10-10T13:34:47.606427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_data[0]","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:34:35.955453Z","iopub.execute_input":"2024-10-10T13:34:35.955943Z","iopub.status.idle":"2024-10-10T13:34:35.965420Z","shell.execute_reply.started":"2024-10-10T13:34:35.955900Z","shell.execute_reply":"2024-10-10T13:34:35.964212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 1\nstart = i * CONFIG.per_row_use\nend = (i + 1 ) * CONFIG.per_row_use\ntmp = df.iloc[start: end, :]\ntmp\n\nper_data = []\nfor col in CONFIG.par_columns:\n    _mean = np.mean(tmp[col].values) # 平均值\n    _max = np.max(tmp[col].values) # 最大值\n    _min = np.min(tmp[col].values) # 最小值\n    _majority, count = stats.mode(data) # 众数\n    _var = np.var(tmp[col].values) # 方差\n    per_data += [_mean, _max, _min, _majority, _var]\n    \nnp.array(per_data)","metadata":{"execution":{"iopub.status.busy":"2024-10-10T13:29:17.485944Z","iopub.execute_input":"2024-10-10T13:29:17.486365Z","iopub.status.idle":"2024-10-10T13:29:17.512743Z","shell.execute_reply.started":"2024-10-10T13:29:17.486325Z","shell.execute_reply":"2024-10-10T13:29:17.511629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}