{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"},{"sourceId":142125,"sourceType":"modelInstanceVersion","modelInstanceId":120396,"modelId":143613},{"sourceId":142914,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":121072,"modelId":144247},{"sourceId":142966,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":121113,"modelId":144284},{"sourceId":144923,"sourceType":"modelInstanceVersion","isSourceIdPinned":true,"modelInstanceId":122835,"modelId":145896}],"dockerImageVersionId":30787,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"simple mlp v1 + focal loss cv:0.22451295967584736 pb: 0.317\n+ +tune alpha(0.1750, 0.2410, 0.2745, 0.3095) cv kappa:  0.23055622601610182([0.2292447606743504, 0.22595353834462328, 0.24695007651697198, 0.17985704499334354, 0.27077570955121977]) pb:","metadata":{}},{"cell_type":"code","source":"import torch \nimport numpy as np\nimport os\nfrom sklearn.metrics import roc_auc_score\nimport pandas as pd\n\nimport torch.nn.functional as F\nfrom torch import nn\nfrom torch.utils.data import DataLoader, Dataset, WeightedRandomSampler\nfrom torch.optim import AdamW\nimport timm\nimport sys\nfrom tqdm import tqdm\n\nfrom PIL import Image\n\nimport albumentations as A\n\nimport math, random\n\nfrom sklearn.model_selection import KFold\nfrom transformers import get_cosine_schedule_with_warmup\n\n\nfrom sklearn.metrics import roc_curve, auc\n\nimport os\nimport random\nfrom tqdm import tqdm\nfrom pathlib import Path\nfrom concurrent.futures import ThreadPoolExecutor\n\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn.metrics import make_scorer\nfrom sklearn.metrics import cohen_kappa_score\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import cross_val_predict\nfrom sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.ensemble import VotingRegressor\nfrom sklearn.impute import SimpleImputer\nfrom scipy.stats import kurtosis\n\nfrom scipy.optimize import minimize\nimport optuna\n\nimport lightgbm as lgb\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\n\nSEED = 42\n\nKAPPA_SCORER = make_scorer(\n    cohen_kappa_score, \n    greater_is_better=True, \n    weights='quadratic',\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:09.478863Z","iopub.execute_input":"2024-10-24T04:25:09.479422Z","iopub.status.idle":"2024-10-24T04:25:41.057249Z","shell.execute_reply.started":"2024-10-24T04:25:09.479357Z","shell.execute_reply":"2024-10-24T04:25:41.056074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_file(filename, dirname):\n    df = pd.read_parquet(os.path.join(dirname, filename, 'part-0.parquet'))\n    df.drop('step', axis=1, inplace=True)\n    \n    return df.describe().values.reshape(-1), filename.split('=')[1]","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:41.059688Z","iopub.execute_input":"2024-10-24T04:25:41.060365Z","iopub.status.idle":"2024-10-24T04:25:41.066849Z","shell.execute_reply.started":"2024-10-24T04:25:41.060303Z","shell.execute_reply":"2024-10-24T04:25:41.065531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_time_series(dirname):\n    ids = os.listdir(dirname)\n    \n    with ThreadPoolExecutor() as executor:\n        results = list(tqdm(executor.map(lambda fname: process_file(fname, dirname), ids), total=len(ids)))\n    \n    stats, indexes = zip(*results)\n    df = pd.DataFrame(stats, columns=[f\"stat_{i}\" for i in range(len(stats[0]))])\n    df['id'] = indexes\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:41.068419Z","iopub.execute_input":"2024-10-24T04:25:41.068808Z","iopub.status.idle":"2024-10-24T04:25:41.082131Z","shell.execute_reply.started":"2024-10-24T04:25:41.068767Z","shell.execute_reply":"2024-10-24T04:25:41.080707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def quadratic_weighted_kappa(estimator, X, y_true):\n    y_pred = estimator.predict(X).round()\n    return cohen_kappa_score(y_true, y_pred, weights='quadratic')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:41.083718Z","iopub.execute_input":"2024-10-24T04:25:41.084211Z","iopub.status.idle":"2024-10-24T04:25:41.094657Z","shell.execute_reply.started":"2024-10-24T04:25:41.084160Z","shell.execute_reply":"2024-10-24T04:25:41.093536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def threshold_rounder(y_pred, thresholds):\n    return np.where(y_pred < thresholds[0], 0,\n                    np.where(y_pred < thresholds[1], 1,\n                             np.where(y_pred < thresholds[2], 2, 3)))","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:41.098589Z","iopub.execute_input":"2024-10-24T04:25:41.099078Z","iopub.status.idle":"2024-10-24T04:25:41.110524Z","shell.execute_reply.started":"2024-10-24T04:25:41.099024Z","shell.execute_reply":"2024-10-24T04:25:41.109246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def eval_preds(thresholds, y_true, y_pred):\n    y_pred = threshold_rounder(y_pred, thresholds)\n    score = cohen_kappa_score(y_true, y_pred, weights='quadratic')\n    return -score","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:41.112131Z","iopub.execute_input":"2024-10-24T04:25:41.112659Z","iopub.status.idle":"2024-10-24T04:25:41.123422Z","shell.execute_reply.started":"2024-10-24T04:25:41.112607Z","shell.execute_reply":"2024-10-24T04:25:41.121946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root = Path('/kaggle/input/child-mind-institute-problematic-internet-use')\ndf_train = pd.read_csv(root / 'train.csv')\ndf_test = pd.read_csv(root / 'test.csv')\ndf_subm = pd.read_csv(root / 'sample_submission.csv', index_col='id')\n\nts_train = load_time_series(root / \"series_train.parquet\")\nts_test = load_time_series(root / \"series_test.parquet\")\n\ntime_series_cols = ts_train.columns.tolist()\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:25:41.124894Z","iopub.execute_input":"2024-10-24T04:25:41.125263Z","iopub.status.idle":"2024-10-24T04:27:18.141212Z","shell.execute_reply.started":"2024-10-24T04:25:41.125210Z","shell.execute_reply":"2024-10-24T04:27:18.139881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"time_series_cols.remove(\"id\")","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.142832Z","iopub.execute_input":"2024-10-24T04:27:18.143284Z","iopub.status.idle":"2024-10-24T04:27:18.149175Z","shell.execute_reply.started":"2024-10-24T04:27:18.143233Z","shell.execute_reply":"2024-10-24T04:27:18.147587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.merge(df_train, ts_train, how=\"left\", on='id')\ndf_test = pd.merge(df_test, ts_test, how=\"left\", on='id')\n\n# df_train = df_train.set_index('id')\n# df_test = df_test.set_index('id')\n\ncat_cols = ['Basic_Demos-Enroll_Season', 'CGAS-Season', 'Physical-Season', 'Fitness_Endurance-Season', 'FGC-Season', 'BIA-Season', 'PAQ_A-Season', 'PAQ_C-Season', 'SDS-Season', 'PreInt_EduHx-Season']\nnum_cols = ['Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-CGAS_Score', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-PAQ_A_Total', 'PAQ_C-PAQ_C_Total', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-computerinternet_hoursday']\ntabular_cols = ['Basic_Demos-Enroll_Season', 'Basic_Demos-Age', 'Basic_Demos-Sex', 'CGAS-Season', 'CGAS-CGAS_Score', 'Physical-Season', 'Physical-BMI', 'Physical-Height', 'Physical-Weight', 'Physical-Waist_Circumference', 'Physical-Diastolic_BP', 'Physical-HeartRate', 'Physical-Systolic_BP', 'Fitness_Endurance-Season', 'Fitness_Endurance-Max_Stage', 'Fitness_Endurance-Time_Mins', 'Fitness_Endurance-Time_Sec', 'FGC-Season', 'FGC-FGC_CU', 'FGC-FGC_CU_Zone', 'FGC-FGC_GSND', 'FGC-FGC_GSND_Zone', 'FGC-FGC_GSD', 'FGC-FGC_GSD_Zone', 'FGC-FGC_PU', 'FGC-FGC_PU_Zone', 'FGC-FGC_SRL', 'FGC-FGC_SRL_Zone', 'FGC-FGC_SRR', 'FGC-FGC_SRR_Zone', 'FGC-FGC_TL', 'FGC-FGC_TL_Zone', 'BIA-Season', 'BIA-BIA_Activity_Level_num', 'BIA-BIA_BMC', 'BIA-BIA_BMI', 'BIA-BIA_BMR', 'BIA-BIA_DEE', 'BIA-BIA_ECW', 'BIA-BIA_FFM', 'BIA-BIA_FFMI', 'BIA-BIA_FMI', 'BIA-BIA_Fat', 'BIA-BIA_Frame_num', 'BIA-BIA_ICW', 'BIA-BIA_LDM', 'BIA-BIA_LST', 'BIA-BIA_SMM', 'BIA-BIA_TBW', 'PAQ_A-Season', 'PAQ_A-PAQ_A_Total', 'PAQ_C-Season', 'PAQ_C-PAQ_C_Total', 'SDS-Season', 'SDS-SDS_Total_Raw', 'SDS-SDS_Total_T', 'PreInt_EduHx-Season', 'PreInt_EduHx-computerinternet_hoursday']\ntarget_col = 'sii'\n\nfeature_cols = tabular_cols + time_series_cols\nnum_cols = num_cols + time_series_cols\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.150792Z","iopub.execute_input":"2024-10-24T04:27:18.151263Z","iopub.status.idle":"2024-10-24T04:27:18.189574Z","shell.execute_reply.started":"2024-10-24T04:27:18.151208Z","shell.execute_reply":"2024-10-24T04:27:18.188429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = df_train.dropna(subset=[target_col])","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.191148Z","iopub.execute_input":"2024-10-24T04:27:18.191653Z","iopub.status.idle":"2024-10-24T04:27:18.202795Z","shell.execute_reply.started":"2024-10-24T04:27:18.191596Z","shell.execute_reply":"2024-10-24T04:27:18.201301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in num_cols:\n    \n#     df_train[col] = (df_train[col] - df_train[col].min()) / (df_train[col].max() - df_train[col].min() +1e-6) ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.204158Z","iopub.execute_input":"2024-10-24T04:27:18.204663Z","iopub.status.idle":"2024-10-24T04:27:18.210806Z","shell.execute_reply.started":"2024-10-24T04:27:18.204607Z","shell.execute_reply":"2024-10-24T04:27:18.209146Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train[time_series_cols].info()","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.212542Z","iopub.execute_input":"2024-10-24T04:27:18.213239Z","iopub.status.idle":"2024-10-24T04:27:18.222667Z","shell.execute_reply.started":"2024-10-24T04:27:18.213177Z","shell.execute_reply":"2024-10-24T04:27:18.221239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imputer = SimpleImputer(\n    strategy='mean',\n)\n\ndf_train[num_cols] = imputer.fit_transform(df_train[num_cols])\ndf_test[num_cols] = imputer.transform(df_test[num_cols])\n\nencoder = OrdinalEncoder(\n    dtype=np.int32,\n    handle_unknown='use_encoded_value',\n    unknown_value=-1,\n    encoded_missing_value=-2,\n)\n\ndf_train[cat_cols] = encoder.fit_transform(df_train[cat_cols])\ndf_train[cat_cols] = df_train[cat_cols].astype('category')\n\ndf_test[cat_cols] = encoder.transform(df_test[cat_cols])\ndf_test[cat_cols] = df_test[cat_cols].astype('category')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.224365Z","iopub.execute_input":"2024-10-24T04:27:18.225349Z","iopub.status.idle":"2024-10-24T04:27:18.338973Z","shell.execute_reply.started":"2024-10-24T04:27:18.225267Z","shell.execute_reply":"2024-10-24T04:27:18.337952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nscaler = StandardScaler()\ndf_train[feature_cols] = scaler.fit_transform(df_train[feature_cols])\ndf_test[feature_cols] = scaler.fit_transform(df_test[feature_cols])","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.343796Z","iopub.execute_input":"2024-10-24T04:27:18.344190Z","iopub.status.idle":"2024-10-24T04:27:18.431253Z","shell.execute_reply.started":"2024-10-24T04:27:18.344149Z","shell.execute_reply":"2024-10-24T04:27:18.429998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_ = df_train['sii'].unique()\ncounts = []\nfor c in class_:\n    count = len(df_train[df_train['sii'] == c])\n    counts.append(count)\n\nfor cls_, c in zip(class_, counts):\n    print(f'class {cls_}, num: {c}')\n\ncounts_ratio = np.array(counts) / sum(counts)\nfor cls_, c in zip(class_, counts_ratio):\n    print(f'class {cls_}, ratio: {c}')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.432926Z","iopub.execute_input":"2024-10-24T04:27:18.433376Z","iopub.status.idle":"2024-10-24T04:27:18.464231Z","shell.execute_reply.started":"2024-10-24T04:27:18.433301Z","shell.execute_reply":"2024-10-24T04:27:18.463107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nNOT_DEBUG = True # True -> run naormally, False -> debug mode, with lesser computing cost\n\nOUTPUT_DIR = f'/kaggle/working/CMI_nn/simple_v1_focalloss'\ndevice = 'cuda:0' if torch.cuda.is_available() else 'cpu'\nN_WORKERS = 32 #os.cpu_count() \nUSE_AMP = True # can change True if using T4 or newer than Ampere\nSEED = 42\n\nsample_SIZE = 256\nIN_CHANS = 1\nN_CLASSES = len(df_train['sii'].unique())\nprint('N_CLASSES: ', N_CLASSES)\n\n\nAUG_PROB = 0.75\n\nN_FOLDS = 5 if NOT_DEBUG else 2\nEPOCHS = 40 if NOT_DEBUG else 2\nMODEL_NAME = 'test_model'\n\nGRAD_ACC = 1\nTGT_BATCH_SIZE = 256\nBATCH_SIZE = TGT_BATCH_SIZE // GRAD_ACC\nMAX_GRAD_NORM = None\nEARLY_STOPPING_EPOCH = 10\n\nLR = 2e-4 * TGT_BATCH_SIZE / 32\nWD = 1e-2\nAUG = True","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.465429Z","iopub.execute_input":"2024-10-24T04:27:18.465751Z","iopub.status.idle":"2024-10-24T04:27:18.475774Z","shell.execute_reply.started":"2024-10-24T04:27:18.465716Z","shell.execute_reply":"2024-10-24T04:27:18.474403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.makedirs(OUTPUT_DIR, exist_ok=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.477686Z","iopub.execute_input":"2024-10-24T04:27:18.478180Z","iopub.status.idle":"2024-10-24T04:27:18.486574Z","shell.execute_reply.started":"2024-10-24T04:27:18.478103Z","shell.execute_reply":"2024-10-24T04:27:18.485383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_random_seed(seed: int = 8620, deterministic: bool = False):\n    \"\"\"Set seeds\"\"\"\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)  # type: ignore\n    torch.backends.cudnn.benchmark = True\n    torch.backends.cudnn.deterministic = deterministic  # type: ignore\n\nset_random_seed(SEED)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.488211Z","iopub.execute_input":"2024-10-24T04:27:18.488835Z","iopub.status.idle":"2024-10-24T04:27:18.506085Z","shell.execute_reply.started":"2024-10-24T04:27:18.488787Z","shell.execute_reply":"2024-10-24T04:27:18.504868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# image_path = '/kaggle/input/isic-2024-challenge/train-image/image'\n# from sklearn.preprocessing import StandardScaler\nclass ISICDataset(Dataset):\n    def __init__(self, df, phase='train', transform=None, return_mean_std=False):\n        self.df = df\n        self.transform = transform\n        self.phase = phase\n        \n#         self.scaled_data = self.scaler.fit_transform(features)\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        if self.phase == 'pred':\n            t = str(self.df.iloc[idx]['id'])\n            data = self.df.iloc[idx][feature_cols].tolist()\n            data = np.array(data).reshape(1, -1)\n            return t, data\n        else:\n            t = str(self.df.iloc[idx]['id'])\n            data = self.df.iloc[idx][feature_cols].tolist()\n            label = int(self.df.iloc[idx]['sii'])\n            data = np.array(data).reshape(1, -1)\n    #         data = (data - data.min()) / (data.max() - data.min() +1e-6) \n            return t, data, label\n\n# transforms_train = A.Compose([\n\n# ])\n\n# transforms_val = A.Compose([\n    \n# ])\n\n\n# from sklearn.preprocessing import StandardScaler\n\n# class ISICDataset(Dataset):\n    \n#     def __init__(self, dataframe, phase='train', transform=None, return_mean_std=False):\n#         # Apply StandardScaler to the input features\n#         self.scaler = StandardScaler()\n#         if 'sii' in dataframe.columns:\n#             self.train = True\n#             features = dataframe.drop(['id', 'sii'], axis=1)  # Drop ID and target column\n#             print(dataframe[features].info())\n            \n\n#             self.targets = dataframe['sii'].values  # Keep target values (sii)\n#         else:\n#             self.train = False\n#             features = dataframe.drop(['id'],axis=1)\n            \n#         self.scaled_data = self.scaler.fit_transform(features)  # Scale features\n            \n\n#     def __len__(self):\n#         return len(self.scaled_data)\n    \n#     def __getitem__(self, idx):\n#         # Return the scaled input features and target value\n#         if self.train:\n#             return 't', torch.tensor(self.scaled_data[idx], dtype=torch.float32), torch.tensor(self.targets[idx], dtype=torch.long)  # Ensure targets are long for classification\n#         else:\n#             return torch.tensor(self.scaled_data[idx], dtype=torch.float32)\n\n\nclass GeM(nn.Module):\n    def __init__(self, p=3, eps=1e-6):\n        super(GeM, self).__init__()\n        self.p = nn.Parameter(torch.ones(1)*p)\n        self.eps = eps\n\n    def forward(self, x):\n        return self.gem(x, p=self.p, eps=self.eps)\n        \n    def gem(self, x, p=3, eps=1e-6):\n        return F.avg_pool1d(x.clamp(min=eps).pow(p), (x.size(-1))).pow(1./p)\n        \n    def __repr__(self):\n        return self.__class__.__name__ + \\\n                '(' + 'p=' + '{:.4f}'.format(self.p.data.tolist()[0]) + \\\n                ', ' + 'eps=' + str(self.eps) + ')'\n\n        \nclass CMIModel(nn.Module):\n    def __init__(self, model_name, in_c=1, n_classes=2, pretrained=True, features_only=False, get_feature=False):\n        super().__init__()\n        self.encoder = nn.Sequential(\n            nn.Linear(in_c, 96),\n            nn.ReLU(),\n            nn.Dropout(0.1),\n            nn.Linear(96, 128),\n            nn.ReLU(),\n            nn.Dropout(0.1),\n            nn.Linear(128, 128),\n            nn.ReLU(),\n        )\n        \n        self.classifier = nn.Sequential(\n            nn.Linear(128, n_classes)\n        )\n\n    def forward(self, x):\n        batch = x.shape[0]\n        x = x.permute(1,0,2)\n        x = self.encoder(x)\n        x = self.classifier(x.reshape(batch, -1))\n        return x\n        ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.508605Z","iopub.execute_input":"2024-10-24T04:27:18.509084Z","iopub.status.idle":"2024-10-24T04:27:18.532364Z","shell.execute_reply.started":"2024-10-24T04:27:18.509029Z","shell.execute_reply":"2024-10-24T04:27:18.531034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df_train['CGAS-Season']","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.533968Z","iopub.execute_input":"2024-10-24T04:27:18.534558Z","iopub.status.idle":"2024-10-24T04:27:18.546477Z","shell.execute_reply.started":"2024-10-24T04:27:18.534512Z","shell.execute_reply":"2024-10-24T04:27:18.545134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for col in feature_cols:\n#     df_train[col] = df_train[col].astype(float)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.548220Z","iopub.execute_input":"2024-10-24T04:27:18.548802Z","iopub.status.idle":"2024-10-24T04:27:18.561627Z","shell.execute_reply.started":"2024-10-24T04:27:18.548730Z","shell.execute_reply":"2024-10-24T04:27:18.560193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"alpha = torch.tensor([0.41, 0.73, 0.86, 0.98])\nF.softmax(alpha)","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.563175Z","iopub.execute_input":"2024-10-24T04:27:18.563598Z","iopub.status.idle":"2024-10-24T04:27:18.679106Z","shell.execute_reply.started":"2024-10-24T04:27:18.563555Z","shell.execute_reply":"2024-10-24T04:27:18.677888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class focal_loss(nn.Module):\n    def __init__(self, alpha=0.5, gamma=2):\n        super().__init__()\n        '''\n        class 2.0, ratio: 0.13815789473684212\n        class 0.0, ratio: 0.5826023391812866\n        class 1.0, ratio: 0.266812865497076\n        class 3.0, ratio: 0.012426900584795321\n        '''\n#         self.alpha=alpha\n        self.gamma=gamma\n        self.alpha = torch.zeros((4,))\n        self.alpha[0] = alpha\n        self.alpha[1:] = 1-alpha\n#         self.alpha = torch.tensor([0.1750, 0.2410, 0.2745, 0.3095])\n#         self.alpha = torch.tensor([0.58, 0.26, 0.13, 0.01])\n#         self.alpha[0] += alpha\n#         self.alpha[1:] += (1-alpha) # α 最终为 [ α, 1-α, 1-α, 1-α, 1-α, ...] size:[num_classes]\n        self.size_average = True\n\n        \n\n    def forward(self, preds, labels):\n        \"\"\"\n        focal_loss损失计算\n        :param preds:   预测类别. size:[B,N,C] or [B,C]    分别对应与检测与分类任务, B批次, N检测框数, C类别数\n        :param labels:  实际类别. size:[B,N] or [B]        [B*N个标签(假设框中有目标)]，[B个标签]\n        :return:\n        \"\"\"\n                \n        #固定类别维度，其余合并(总检测框数或总批次数)，preds.size(-1)是最后一个维度\n        preds = preds.view(-1,preds.size(-1))\n        self.alpha = self.alpha.to(preds.device)\n        \n        #使用log_softmax解决溢出问题，方便交叉熵计算而不用考虑值域\n        preds_logsoft = F.log_softmax(preds, dim=1) \n        \n     \t#log_softmax是softmax+log运算，那再exp就算回去了变成softmax\n        preds_softmax = torch.exp(preds_logsoft)    \n   \n        # 这部分实现nll_loss ( crossentropy = log_softmax + nll)\n        preds_softmax = preds_softmax.gather(1,labels.view(-1,1)) \n        preds_logsoft = preds_logsoft.gather(1,labels.view(-1,1))\n\n        self.alpha = self.alpha.gather(0,labels.view(-1))\n\n        # torch.pow((1-preds_softmax), self.gamma) 为focal loss中 (1-pt)**γ\n        \n        #torch.mul 矩阵对应位置相乘，大小一致\n        loss = -torch.mul(torch.pow((1-preds_softmax), self.gamma), preds_logsoft) \n    \n        #torch.t()求转置\n        loss = torch.mul(self.alpha, loss.t())\n        #print(loss.size()) [1,5]\n        \n        if self.size_average:\n            loss = loss.mean()\n        else:\n            loss = loss.sum()\n       \n        return loss\n","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.680665Z","iopub.execute_input":"2024-10-24T04:27:18.681603Z","iopub.status.idle":"2024-10-24T04:27:18.694799Z","shell.execute_reply.started":"2024-10-24T04:27:18.681549Z","shell.execute_reply":"2024-10-24T04:27:18.693440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfrom collections import OrderedDict\nautocast = torch.cuda.amp.autocast(USE_AMP) # you can use with T4 gpu. or newer\nscaler = torch.cuda.amp.GradScaler(enabled=USE_AMP, init_scale=4096)\n\nfrom sklearn.model_selection import StratifiedGroupKFold\n\n# skf = StratifiedGroupKFold(n_splits=N_FOLDS, shuffle=True, random_state=42)\nskf = StratifiedKFold(5, shuffle=True, random_state=SEED)\n\nalphas = [0.05, 0.15, 0.25, 0.35, 0.45]\ngamas = [1.5, 2, 2.3, 2.5]\n\ntrain=False\nif train:\n    for alpha in alphas:\n        for gama in gamas:\n            os.makedirs(f'{OUTPUT_DIR}_{alpha}_{gama}', exist_ok=True)\n            best_kappas = []\n            for fold, (trn_idx, test_idx) in enumerate(skf.split(df_train[feature_cols],df_train['sii'])):\n\n                train_ds = ISICDataset(df_train.iloc[trn_idx], phase='train', transform=None)\n                train_dl = DataLoader(\n                            train_ds,\n                            batch_size=BATCH_SIZE,\n                            # sampler=sampler,\n                            shuffle=True,\n                            pin_memory=True,\n                            drop_last=True,\n                            num_workers=N_WORKERS\n                            )\n\n                valid_ds = ISICDataset(df_train.iloc[test_idx], phase='valid', transform=None)\n                valid_dl = DataLoader(\n                            valid_ds,\n                            batch_size=BATCH_SIZE,\n                            # sampler=sampler,\n                            shuffle=False,\n                            pin_memory=True,\n                            drop_last=False,\n                            num_workers=N_WORKERS\n                            )\n\n\n                model = CMIModel('', len(feature_cols), N_CLASSES, pretrained=True)\n        #         model = CMIModel_1('', len(feature_cols), N_CLASSES, pretrained=True)\n            #     print(model, device)\n                model.to(device)\n\n                optimizer = AdamW(model.parameters(), lr=LR, weight_decay=WD)\n\n                warmup_steps = len(train_dl) // GRAD_ACC\n                num_total_steps = EPOCHS * len(train_dl) // GRAD_ACC\n                num_cycles = 0.475\n                scheduler = get_cosine_schedule_with_warmup(optimizer,\n                                                            num_warmup_steps=warmup_steps,\n                                                            num_training_steps=num_total_steps,\n                                                            num_cycles=num_cycles)\n\n                weights = torch.tensor([1.0, 10.0])\n                criterion = focal_loss(alpha, gama) #nn.CrossEntropyLoss()#focal_loss_from_logits()\n                # criterion = binary_ce_loss_from_logits()\n                criterion2 = criterion #nn.BCELoss(weight=weights)\n                correct = 0\n                total = 0\n                best_cel = 1.2\n                best_kappa = 0.002\n                es_step = 0\n\n                for epoch in range(1, EPOCHS+1):\n                    print(f'start epoch {epoch}')\n                    model.train()\n                    total_loss = 0\n                    with tqdm(train_dl, leave=True) as pbar:\n                        optimizer.zero_grad()\n                        for idx, (_, x, t) in enumerate(pbar):  \n                            x = x.to(device, dtype=torch.float)\n                            t = t.to(device, dtype=torch.long)\n            #                 print(t)\n                            with autocast:\n                                y = model(x)\n                                loss = criterion(y, t)    \n                                total_loss += loss.item()\n                                if GRAD_ACC > 1:\n                                    loss = loss / GRAD_ACC\n\n                            if not math.isfinite(loss):\n                                print(f\"Loss is {loss}, stopping training\")\n                                sys.exit(1)\n\n                            pbar.set_postfix(\n                                OrderedDict(\n                                    loss=f'{loss.item()*GRAD_ACC:.6f}',\n                                    lr=f'{optimizer.param_groups[0][\"lr\"]:.3e}'\n                                )\n                            )\n                            scaler.scale(loss).backward()\n\n                            torch.nn.utils.clip_grad_norm_(model.parameters(), MAX_GRAD_NORM or 1e9)\n\n                            if (idx + 1) % GRAD_ACC == 0:\n                                scaler.step(optimizer)\n                                scaler.update()\n                                optimizer.zero_grad()\n                                if scheduler is not None:\n                                    scheduler.step()                    \n\n                    train_loss = total_loss/len(train_dl)\n                    print(f'train_loss:{train_loss:.6f}')\n\n                    model.eval()\n                    total_loss = 0\n                    labels = []\n                    all_preds = []\n                    with tqdm(valid_dl, leave=True) as pbar:\n                        with torch.no_grad():\n                            for idx, (_, x, t) in enumerate(pbar):  \n                                x = x.to(device, dtype=torch.float)\n                                t = t.to(device, dtype=torch.long)\n                                with autocast:\n                                    y = model(x)\n                                    loss = criterion(y, t)    \n                                    total_loss += loss.item()\n                                    pred_y = torch.argmax(y, dim=1)\n                                    pred_y = pred_y.float().cpu().detach().numpy()\n                                    gt_y = t.float().cpu().detach().numpy()\n\n                                    labels.append(gt_y)\n                                    all_preds.append(pred_y)\n\n                    labels = np.concatenate(labels, axis=0)\n                    all_preds = np.concatenate(all_preds, axis=0)\n                    kappa = cohen_kappa_score(all_preds.reshape(-1,), labels.reshape(-1,))\n\n                    valid_loss = total_loss/len(valid_dl)\n                    print(f'valid_loss:{valid_loss:.6f}, kappa: {kappa:.6f}')\n\n                    if best_kappa < kappa:\n                        best_kappa = kappa\n                        print(f'best kappa {kappa:.6f} in epoch {epoch}')\n                        \n                        fname = f'{OUTPUT_DIR}_{alpha}_{gama}/best_kappa_model_fold-{fold}.pt'\n                        torch.save(model.state_dict(), fname)\n                    else:\n                        es_step += 1\n                    if es_step >= EARLY_STOPPING_EPOCH:\n                        break\n\n                best_kappas.append(best_kappa)\n\n            print(best_kappas)\n            print(f'cv kappa:  {sum(best_kappas)/len(best_kappas)}')","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.696554Z","iopub.execute_input":"2024-10-24T04:27:18.696987Z","iopub.status.idle":"2024-10-24T04:27:18.732497Z","shell.execute_reply.started":"2024-10-24T04:27:18.696947Z","shell.execute_reply":"2024-10-24T04:27:18.731161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !tar -czvf /kaggle/working/CMI_nn.tar /kaggle/working/CMI_nn ","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.733974Z","iopub.execute_input":"2024-10-24T04:27:18.734474Z","iopub.status.idle":"2024-10-24T04:27:18.742829Z","shell.execute_reply.started":"2024-10-24T04:27:18.734420Z","shell.execute_reply":"2024-10-24T04:27:18.741624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_pred = False","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.744274Z","iopub.execute_input":"2024-10-24T04:27:18.744746Z","iopub.status.idle":"2024-10-24T04:27:18.752663Z","shell.execute_reply.started":"2024-10-24T04:27:18.744692Z","shell.execute_reply":"2024-10-24T04:27:18.751519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for fold, (trn_idx, test_idx) in enumerate(skf.split(df_train[feature_cols],df_train['sii'])):\n# df_subm = pd.DataFrame()\n# test_path = '/kaggle/input/example/pytorch/default/1/kappa_model_simple_v1'\n# test_path = '/kaggle/input/simple_v1focal-loss/pytorch/default/1/kappa_model_simple_v1_focalloss'\n\n\ntest_path = '/kaggle/input/simple_v1focallosstune_alpha/pytorch/default/1/kappa_model_simple_v1_focalloss_tunealpha'\ntest_paths = [\n    '/kaggle/input/example/pytorch/default/1/kappa_model_simple_v1',\n    '/kaggle/input/simple_v1focal-loss/pytorch/default/1/kappa_model_simple_v1_focalloss',\n    '/kaggle/input/simple_v1focallosstune_alpha/pytorch/default/1/kappa_model_simple_v1_focalloss_tunealpha'\n]\n\npath = '/kaggle/input/cmi_nn_archive/pytorch/default/1/CMI_nn'\nfor p in os.listdir(path):\n    test_paths += [os.path.join(path, p)]\n\nprint(len(test_paths))\n\nif not mean_pred:\n    for fold in range(5):\n        for idx_, test_path in enumerate(test_paths):\n            valid_ds = ISICDataset(df_test, phase='pred', transform=None)\n            valid_dl = DataLoader(\n                        valid_ds,\n                        batch_size=1,\n                        shuffle=False,\n                        pin_memory=True,\n                        drop_last=False,\n                        num_workers=N_WORKERS\n                        )\n\n\n            model = CMIModel('', len(feature_cols), N_CLASSES, pretrained=True)\n\n            fname = f'{test_path}/best_kappa_model_fold-{fold}.pt'\n            model.load_state_dict(torch.load(fname))\n            model.eval()\n        #     model.half()\n            model.to(device)\n\n            ids = []\n            all_preds = []\n            print(test_path, fold)\n            with tqdm(valid_dl, leave=True) as pbar:\n                with torch.no_grad():\n                    for idx, (id_, x) in enumerate(pbar):  \n                        x = x.to(device, dtype=torch.float)\n                        id_ = id_[0]\n                        ids.append(id_)\n                        with autocast:\n                            y = model(x)\n                            pred_y = torch.argmax(y, dim=1)\n                            pred_y = pred_y.float().cpu().detach().numpy()\n\n                            all_preds.append(pred_y)\n    #     if fold == 0:\n    #         df_subm['id'] = ids\n            all_preds = np.concatenate(all_preds, axis=0)\n            all_preds = np.asarray(all_preds, dtype=int) \n            df_subm[f'si_{fold}_{idx_}'] = all_preds\n\ndf_subm[target_col] = df_subm.loc[:,[f'si_{i}_{j}' for i in range(5) for j in range(len(test_paths))]].mode(axis=1)[0]\nfor i in range(5):\n    for j in range(len(test_paths)):\n        del df_subm[f'si_{i}_{j}']\n\ndf_subm[target_col] = df_subm[target_col].astype(int)\ndf_subm.to_csv('submission.csv')\ndf_subm","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:27:18.754278Z","iopub.execute_input":"2024-10-24T04:27:18.754811Z","iopub.status.idle":"2024-10-24T04:29:56.718123Z","shell.execute_reply.started":"2024-10-24T04:27:18.754764Z","shell.execute_reply":"2024-10-24T04:29:56.716311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_subm","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:29:56.720968Z","iopub.execute_input":"2024-10-24T04:29:56.721464Z","iopub.status.idle":"2024-10-24T04:29:56.735880Z","shell.execute_reply.started":"2024-10-24T04:29:56.721414Z","shell.execute_reply":"2024-10-24T04:29:56.734623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_path = '/kaggle/input/simple_v1focallosstune_alpha/pytorch/default/1/kappa_model_simple_v1_focalloss_tunealpha'\n# test_paths = [\n#     '/kaggle/input/example/pytorch/default/1/kappa_model_simple_v1',\n#     '/kaggle/input/simple_v1focal-loss/pytorch/default/1/kappa_model_simple_v1_focalloss',\n#     '/kaggle/input/simple_v1focallosstune_alpha/pytorch/default/1/kappa_model_simple_v1_focalloss_tunealpha'\n# ]\n\n# if mean_pred:\n#     final_preds = []\n#     for fold in range(5):\n#         for idx_, test_path in enumerate(test_paths):\n#             valid_ds = ISICDataset(df_test, phase='pred', transform=None)\n#             valid_dl = DataLoader(\n#                         valid_ds,\n#                         batch_size=1,\n#                         shuffle=False,\n#                         pin_memory=True,\n#                         drop_last=False,\n#                         num_workers=N_WORKERS\n#                         )\n\n\n#             model = CMIModel('', len(feature_cols), N_CLASSES, pretrained=True)\n\n#             fname = f'{test_path}/best_kappa_model_fold-{fold}.pt'\n#             model.load_state_dict(torch.load(fname))\n#             model.eval()\n#         #     model.half()\n#             model.to(device)\n\n#             ids = []\n#             all_preds = []\n#             with tqdm(valid_dl, leave=True) as pbar:\n#                 with torch.no_grad():\n#                     for idx, (id_, x) in enumerate(pbar):  \n#                         x = x.to(device, dtype=torch.float)\n#                         id_ = id_[0]\n#                         ids.append(id_)\n#                         with autocast:\n#                             y = model(x)\n#                             pred_y = y #torch.argmax(y, dim=1)\n#                             pred_y = pred_y.float().cpu().detach().numpy()\n\n#                             all_preds.append(pred_y)\n\n#             all_preds = np.concatenate(all_preds, axis=0)\n#             all_preds = np.asarray(all_preds, dtype=int) \n\n#             final_preds.append(all_preds)\n\n\n#     get_mean_preds = np.zeros_like(final_preds[0], dtype=np.float64)\n#     for i in range(len(final_preds)):\n#         get_mean_preds += final_preds[i] / len(final_preds)\n\n#     df_subm[target_col] = np.argmax(get_mean_preds, axis=1)\n#     df_subm[target_col] = df_subm[target_col].astype(int)\n#     df_subm.to_csv('submission.csv')\n#     df_subm","metadata":{"execution":{"iopub.status.busy":"2024-10-24T04:29:56.737485Z","iopub.execute_input":"2024-10-24T04:29:56.737891Z","iopub.status.idle":"2024-10-24T04:29:57.811102Z","shell.execute_reply.started":"2024-10-24T04:29:56.737837Z","shell.execute_reply":"2024-10-24T04:29:57.809557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}