{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":10384,"databundleVersionId":120379}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Cell 1: Import Libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report, f1_score\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom sklearn.ensemble import RandomForestClassifier, ExtraTreesClassifier, VotingClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.base import BaseEstimator, ClassifierMixin\nimport lightgbm as lgb\nfrom scipy import stats\nfrom tqdm import tqdm\nimport time\nimport warnings\nwarnings.filterwarnings('ignore')\n\npd.set_option('display.max_columns', 100)\nplt.style.use('seaborn-v0_8-darkgrid')\n\nprint(\"Libraries imported successfully\")","metadata":{"_uuid":"cb07a028-b3fb-43cd-aa34-25efc222fb51","_cell_guid":"734869e9-c4a6-4a88-8602-df7dfa22290b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:35:45.723459Z","iopub.execute_input":"2025-11-30T23:35:45.724335Z","iopub.status.idle":"2025-11-30T23:35:45.730837Z","shell.execute_reply.started":"2025-11-30T23:35:45.724301Z","shell.execute_reply":"2025-11-30T23:35:45.730185Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 2: Load Data\nprint(\"Loading PLAsTiCC dataset...\")\ntrain_meta = pd.read_csv('../input/PLAsTiCC-2018/training_set_metadata.csv')\ntrain = pd.read_csv('../input/PLAsTiCC-2018/training_set.csv')\n\nSAMPLE_SIZE = None\n\nif SAMPLE_SIZE:\n    train_ids = train['object_id'].unique()[:SAMPLE_SIZE//100]\n    train = train[train['object_id'].isin(train_ids)]\n    train_meta = train_meta[train_meta['object_id'].isin(train_ids)]\n\nprint(f\"Train shape: {train.shape}\")\nprint(f\"Train meta shape: {train_meta.shape}\")\nprint(f\"Unique objects: {train['object_id'].nunique()}\")\nprint(f\"Number of classes: {train_meta['target'].nunique()}\")","metadata":{"_uuid":"e38444bd-7e9e-4e75-b893-7d1757dee4fc","_cell_guid":"dd69ac29-2c43-4cae-8369-4168412b865d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:35:45.732011Z","iopub.execute_input":"2025-11-30T23:35:45.732221Z","iopub.status.idle":"2025-11-30T23:35:46.463607Z","shell.execute_reply.started":"2025-11-30T23:35:45.732204Z","shell.execute_reply":"2025-11-30T23:35:46.462823Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 3: Feature Engineering Functions\ndef extract_features(df):\n    print(\"Extracting statistical features per passband...\")\n    \n    features = df.groupby(['object_id', 'passband']).agg({\n        'flux': [\n            'mean', 'std', 'min', 'max', 'skew',\n            lambda x: stats.kurtosis(x),\n            lambda x: np.percentile(x, 25),\n            lambda x: np.percentile(x, 75),\n            lambda x: np.percentile(x, 90),\n            lambda x: x.max() - x.min(),\n            lambda x: np.median(np.abs(x - np.median(x))),\n            lambda x: np.sum(x > 0) / len(x) if len(x) > 0 else 0,\n            lambda x: np.sum(x > x.mean()) / len(x) if len(x) > 0 else 0,\n        ],\n        'flux_err': [\n            'mean', 'std', 'min', 'max',\n            lambda x: np.mean(x) / (np.std(x) + 1e-8) if np.std(x) > 0 else 0\n        ],\n        'detected': ['sum', 'mean'],\n        'mjd': [\n            'min', 'max', 'count',\n            lambda x: np.std(np.diff(sorted(x))) if len(x) > 1 else 0\n        ]\n    })\n    \n    new_columns = []\n    for col in features.columns:\n        if col[1] == '<lambda>':\n            if col[0] == 'flux':\n                lambda_names = ['kurtosis', 'q1', 'q3', 'p90', 'range', 'mad', 'pos_frac', 'above_mean_frac']\n                lambda_index = len([c for c in new_columns if c.startswith(f'{col[0]}_')])\n                if lambda_index < len(lambda_names):\n                    new_columns.append(f'{col[0]}_{lambda_names[lambda_index]}')\n                else:\n                    new_columns.append(f'{col[0]}_lambda{lambda_index}')\n            elif col[0] == 'flux_err':\n                new_columns.append(f'{col[0]}_snr')\n            elif col[0] == 'mjd':\n                new_columns.append(f'{col[0]}_gap_std')\n        else:\n            new_columns.append(f'{col[0]}_{col[1]}')\n    \n    features.columns = new_columns\n    features = features.reset_index()\n    \n    features_pivot = features.pivot_table(\n        index='object_id',\n        columns='passband',\n        values=[col for col in features.columns if col not in ['object_id', 'passband']]\n    )\n    \n    features_pivot.columns = ['_'.join(map(str, col)) for col in features_pivot.columns]\n    features_pivot = features_pivot.reset_index()\n    \n    return features_pivot\n\ndef add_time_features(df):\n    print(\"Extracting time-based features...\")\n    \n    time_features = df.groupby('object_id').agg({\n        'mjd': lambda x: x.max() - x.min(),\n        'flux': [\n            lambda x: np.sum(np.abs(np.diff(x))),\n            lambda x: len(x[x > np.median(x)]) / len(x) if len(x) > 0 else 0,\n            lambda x: np.mean(np.abs(np.diff(x))) if len(x) > 1 else 0\n        ],\n        'detected': lambda x: x.sum() / len(x)\n    })\n    \n    time_features.columns = ['duration', 'total_variation', 'above_median_frac',\n                             'mean_flux_change', 'detection_rate']\n    time_features = time_features.reset_index()\n    \n    return time_features\n\ndef add_peak_features(df):\n    print(\"Extracting peak and rise/fall features...\")\n    peak_features = []\n    \n    for obj_id in tqdm(df['object_id'].unique(), desc=\"Peak features\"):\n        obj_data = df[df['object_id'] == obj_id]\n        features_dict = {'object_id': obj_id}\n        \n        for pb in range(6):\n            pb_data = obj_data[obj_data['passband'] == pb]\n            \n            if len(pb_data) > 3:\n                peak_idx = pb_data['flux'].idxmax()\n                features_dict[f'peak_flux_{pb}'] = pb_data.loc[peak_idx, 'flux']\n                features_dict[f'peak_mjd_{pb}'] = pb_data.loc[peak_idx, 'mjd']\n                features_dict[f'time_to_peak_{pb}'] = (\n                    pb_data.loc[peak_idx, 'mjd'] - pb_data['mjd'].min()\n                )\n                \n                peak_time = pb_data.loc[peak_idx, 'mjd']\n                before_peak = pb_data[pb_data['mjd'] < peak_time]\n                after_peak = pb_data[pb_data['mjd'] > peak_time]\n                \n                if len(before_peak) > 1:\n                    rise_rate = (pb_data.loc[peak_idx, 'flux'] - before_peak['flux'].min()) / (\n                        peak_time - before_peak['mjd'].min() + 1e-8)\n                    features_dict[f'rise_rate_{pb}'] = rise_rate\n                else:\n                    features_dict[f'rise_rate_{pb}'] = 0\n                    \n                if len(after_peak) > 1:\n                    fall_rate = (pb_data.loc[peak_idx, 'flux'] - after_peak['flux'].min()) / (\n                        after_peak['mjd'].max() - peak_time + 1e-8)\n                    features_dict[f'fall_rate_{pb}'] = fall_rate\n                else:\n                    features_dict[f'fall_rate_{pb}'] = 0\n            else:\n                features_dict[f'peak_flux_{pb}'] = np.nan\n                features_dict[f'peak_mjd_{pb}'] = np.nan\n                features_dict[f'time_to_peak_{pb}'] = np.nan\n                features_dict[f'rise_rate_{pb}'] = np.nan\n                features_dict[f'fall_rate_{pb}'] = np.nan\n        \n        peak_features.append(features_dict)\n    \n    return pd.DataFrame(peak_features)\n\ndef add_color_features(df):\n    print(\"Extracting color features...\")\n    color_features = []\n    \n    for obj_id in tqdm(df['object_id'].unique(), desc=\"Color features\"):\n        obj_data = df[df['object_id'] == obj_id]\n        features_dict = {'object_id': obj_id}\n        \n        for pb1 in range(5):\n            for pb2 in range(pb1+1, 6):\n                pb1_data = obj_data[obj_data['passband'] == pb1]\n                pb2_data = obj_data[obj_data['passband'] == pb2]\n                \n                if len(pb1_data) > 0 and len(pb2_data) > 0:\n                    pb1_flux = pb1_data['flux'].mean()\n                    pb2_flux = pb2_data['flux'].mean()\n                    \n                    if pb1_flux > 0 and pb2_flux > 0:\n                        color = -2.5 * np.log10(pb1_flux / pb2_flux)\n                        features_dict[f'color_{pb1}_{pb2}'] = color\n                    else:\n                        features_dict[f'color_{pb1}_{pb2}'] = 0\n                else:\n                    features_dict[f'color_{pb1}_{pb2}'] = np.nan\n                    \n        color_features.append(features_dict)\n    \n    return pd.DataFrame(color_features)\n\ndef add_advanced_features(df):\n    print(\"Extracting advanced features...\")\n    adv_features = []\n    \n    for obj_id in tqdm(df['object_id'].unique(), desc=\"Advanced features\"):\n        obj_data = df[df['object_id'] == obj_id]\n        features_dict = {'object_id': obj_id}\n        \n        for pb in range(6):\n            pb_data = obj_data[obj_data['passband'] == pb].sort_values('mjd')\n            if len(pb_data) > 10:\n                flux = pb_data['flux'].values\n                flux = flux - np.mean(flux)\n                fft = np.fft.fft(flux)\n                freqs = np.fft.fftfreq(len(flux))\n                \n                pos_mask = freqs > 0\n                fft_pos = np.abs(fft[pos_mask])\n                \n                if len(fft_pos) > 0:\n                    features_dict[f'fft_max_freq_{pb}'] = fft_pos.max()\n                    features_dict[f'fft_mean_amp_{pb}'] = fft_pos.mean()\n                    features_dict[f'fft_std_amp_{pb}'] = fft_pos.std()\n                else:\n                    features_dict[f'fft_max_freq_{pb}'] = 0\n                    features_dict[f'fft_mean_amp_{pb}'] = 0\n                    features_dict[f'fft_std_amp_{pb}'] = 0\n            else:\n                features_dict[f'fft_max_freq_{pb}'] = np.nan\n                features_dict[f'fft_mean_amp_{pb}'] = np.nan\n                features_dict[f'fft_std_amp_{pb}'] = np.nan\n        \n        for pb1 in range(6):\n            for pb2 in range(pb1+1, 6):\n                pb1_flux = obj_data[obj_data['passband'] == pb1]['flux'].mean()\n                pb2_flux = obj_data[obj_data['passband'] == pb2]['flux'].mean()\n                \n                if pb2_flux != 0 and not np.isnan(pb2_flux):\n                    features_dict[f'flux_ratio_{pb1}_{pb2}'] = pb1_flux / pb2_flux\n                else:\n                    features_dict[f'flux_ratio_{pb1}_{pb2}'] = 0\n        \n        total_flux = obj_data['flux'].sum()\n        if total_flux > 0:\n            features_dict['weighted_mean_time'] = (obj_data['flux'] * obj_data['mjd']).sum() / total_flux\n        else:\n            features_dict['weighted_mean_time'] = obj_data['mjd'].mean()\n            \n        for pb in range(6):\n            pb_data = obj_data[obj_data['passband'] == pb]['flux']\n            if len(pb_data) > 0:\n                features_dict[f'flux_5th_percentile_{pb}'] = np.percentile(pb_data, 5)\n                features_dict[f'flux_95th_percentile_{pb}'] = np.percentile(pb_data, 95)\n                features_dict[f'flux_iqr_{pb}'] = np.percentile(pb_data, 75) - np.percentile(pb_data, 25)\n            else:\n                features_dict[f'flux_5th_percentile_{pb}'] = np.nan\n                features_dict[f'flux_95th_percentile_{pb}'] = np.nan\n                features_dict[f'flux_iqr_{pb}'] = np.nan\n        \n        for pb in range(6):\n            pb_data = obj_data[obj_data['passband'] == pb].sort_values('mjd')\n            if len(pb_data) > 2:\n                flux_diff = np.diff(pb_data['flux'].values)\n                time_diff = np.diff(pb_data['mjd'].values)\n                \n                time_diff[time_diff == 0] = 1e-8\n                flux_derivative = flux_diff / time_diff\n                \n                features_dict[f'flux_deriv_max_{pb}'] = flux_derivative.max()\n                features_dict[f'flux_deriv_min_{pb}'] = flux_derivative.min()\n                features_dict[f'flux_deriv_std_{pb}'] = flux_derivative.std()\n            else:\n                features_dict[f'flux_deriv_max_{pb}'] = np.nan\n                features_dict[f'flux_deriv_min_{pb}'] = np.nan\n                features_dict[f'flux_deriv_std_{pb}'] = np.nan\n                \n        adv_features.append(features_dict)\n    \n    return pd.DataFrame(adv_features)","metadata":{"_uuid":"bf9e3c52-e19c-433a-919b-e10e81fdd5d8","_cell_guid":"6a7318b3-b4aa-4ad0-b16d-c520df3328e5","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:35:46.464832Z","iopub.execute_input":"2025-11-30T23:35:46.465085Z","iopub.status.idle":"2025-11-30T23:35:46.490708Z","shell.execute_reply.started":"2025-11-30T23:35:46.465067Z","shell.execute_reply":"2025-11-30T23:35:46.490019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 4: Data Augmentation for Rare Classes\nprint(\"Performing data augmentation for rare classes...\")\n\ntrain_original = train.copy()\ntrain_meta_original = train_meta.copy()\n\nclass_counts = train_meta['target'].value_counts()\nrare_classes = class_counts[class_counts < 100].index.tolist()\n\nprint(f\"Rare classes to augment: {rare_classes}\")\nprint(f\"Class distribution before augmentation:\")\nfor cls in rare_classes:\n    print(f\"  Class {cls}: {class_counts[cls]} samples\")\n\ndef advanced_augment_light_curves(df, meta_df, target_classes, n_augment=10):\n    augmented_data = []\n    augmented_meta = []\n    \n    target_objects = meta_df[meta_df['target'].isin(target_classes)]['object_id'].values\n    \n    print(f\"Augmenting {len(target_objects)} objects\")\n    \n    for obj_id in tqdm(target_objects, desc=\"Augmenting\"):\n        obj_data = df[df['object_id'] == obj_id].copy()\n        obj_meta = meta_df[meta_df['object_id'] == obj_id].copy()\n        \n        peak_flux_dict = {}\n        for pb in range(6):\n            pb_data = obj_data[obj_data['passband'] == pb]\n            if len(pb_data) > 0:\n                peak_flux_dict[pb] = pb_data['flux'].max()\n        \n        for i in range(n_augment):\n            aug_data = obj_data.copy()\n            aug_meta = obj_meta.copy()\n            \n            for pb in range(6):\n                pb_mask = aug_data['passband'] == pb\n                pb_data = aug_data[pb_mask].copy()\n                \n                if len(pb_data) > 3:\n                    pb_data = pb_data.sort_values('mjd')\n                    \n                    noise = np.random.randn(len(pb_data))\n                    for j in range(1, len(noise)):\n                        noise[j] = 0.6 * noise[j-1] + 0.4 * noise[j]\n                    \n                    noise_scaled = noise * pb_data['flux_err'].values * 0.2\n                    pb_data['flux'] += noise_scaled\n                    \n                    aug_data.loc[pb_mask, 'flux'] = pb_data['flux'].values\n            \n            time_factor = np.random.uniform(0.95, 1.05)\n            min_mjd = aug_data['mjd'].min()\n            aug_data['mjd'] = min_mjd + (aug_data['mjd'] - min_mjd) * time_factor\n            \n            extinction = np.random.uniform(0, 0.2)\n            passband_extinction = {\n                0: 1.0 - extinction * 0.4,\n                1: 1.0 - extinction * 0.3,\n                2: 1.0 - extinction * 0.2,\n                3: 1.0 - extinction * 0.15,\n                4: 1.0 - extinction * 0.1,\n                5: 1.0 - extinction * 0.05\n            }\n            \n            for pb, factor in passband_extinction.items():\n                mask = aug_data['passband'] == pb\n                aug_data.loc[mask, 'flux'] *= factor\n                aug_data.loc[mask, 'flux_err'] *= factor\n            \n            phase_shift = np.random.uniform(-3, 3)\n            aug_data['mjd'] += phase_shift\n            \n            for pb in peak_flux_dict:\n                pb_mask = aug_data['passband'] == pb\n                min_allowed = -0.2 * peak_flux_dict[pb]\n                aug_data.loc[pb_mask & (aug_data['flux'] < min_allowed), 'flux'] = min_allowed\n            \n            new_obj_id = obj_id * 10000 + (i + 1)\n            aug_data['object_id'] = new_obj_id\n            aug_meta['object_id'] = new_obj_id\n            \n            if 'hostgal_photoz' in aug_meta.columns:\n                aug_meta['hostgal_photoz'] += np.random.normal(0, 0.02)\n                aug_meta['hostgal_photoz'] = aug_meta['hostgal_photoz'].clip(0, 3)\n            \n            augmented_data.append(aug_data)\n            augmented_meta.append(aug_meta)\n    \n    if len(augmented_data) > 0:\n        return pd.concat(augmented_data), pd.concat(augmented_meta)\n    else:\n        return pd.DataFrame(), pd.DataFrame()\n\nif len(rare_classes) > 0:\n    aug_train, aug_meta = advanced_augment_light_curves(\n        train, train_meta, rare_classes, n_augment=10\n    )\n    \n    if len(aug_train) > 0:\n        train = pd.concat([train, aug_train], ignore_index=True)\n        train_meta = pd.concat([train_meta, aug_meta], ignore_index=True)\n        \n        print(f\"\\nOriginal samples: {len(train_original)}\")\n        print(f\"After augmentation: {len(train)}\")\n        print(f\"Increase: +{len(train) - len(train_original)} samples\")","metadata":{"_uuid":"16839cad-23fb-4c67-ade5-f5cc383dd22b","_cell_guid":"b71b03fb-8969-4091-a9d5-3de47b66b8fb","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:35:46.491471Z","iopub.execute_input":"2025-11-30T23:35:46.491670Z","iopub.status.idle":"2025-11-30T23:35:51.688183Z","shell.execute_reply.started":"2025-11-30T23:35:46.491655Z","shell.execute_reply":"2025-11-30T23:35:51.687611Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 5: Label Encoding and Feature Extraction\nprint(\"\\nLabel Encoding...\")\nprint(f\"Original unique classes: {sorted(train_meta['target'].unique())}\")\n\nclass_mapping = {c: i for i, c in enumerate(sorted(train_meta['target'].unique()))}\nreverse_mapping = {i: c for c, i in class_mapping.items()}\n\nprint(\"\\nClass mapping:\")\nfor original, encoded in class_mapping.items():\n    print(f\"Class {original} -> {encoded}\")\n\ntrain_meta['target_original'] = train_meta['target'].copy()\ntrain_meta['target'] = train_meta['target'].map(class_mapping)\n\nprint(\"\\nFeature Extraction...\")\n\nstat_features = extract_features(train)\nprint(f\"Statistical features: {stat_features.shape}\")\n\ntime_features = add_time_features(train)\nprint(f\"Time features: {time_features.shape}\")\n\npeak_features = add_peak_features(train)\nprint(f\"Peak features: {peak_features.shape}\")\n\ncolor_features = add_color_features(train)\nprint(f\"Color features: {color_features.shape}\")\n\nadvanced_features = add_advanced_features(train)\nprint(f\"Advanced features: {advanced_features.shape}\")\n\nprint(\"\\nMerging all features...\")\nfeatures = stat_features\nfeatures = features.merge(time_features, on='object_id', how='left')\nfeatures = features.merge(peak_features, on='object_id', how='left')\nfeatures = features.merge(color_features, on='object_id', how='left')\nfeatures = features.merge(advanced_features, on='object_id', how='left')\nfeatures = features.merge(train_meta, on='object_id', how='left')\n\nduplicate_cols = features.columns[features.columns.duplicated()].tolist()\nif duplicate_cols:\n    print(f\"Warning: Found duplicate columns: {duplicate_cols}\")\n    features = features.loc[:, ~features.columns.duplicated()]\n\nprint(f\"\\nMissing values before filling: {features.isnull().sum().sum()}\")\nfeatures = features.fillna(0)\n\nprint(f\"Final features shape: {features.shape}\")\nprint(f\"Target range: {features['target'].min()} to {features['target'].max()}\")","metadata":{"_uuid":"d4058b4d-b128-42f6-8439-dfc9154e3e67","_cell_guid":"2c286207-80d4-4b72-9bf9-87066d74c6e5","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:35:51.689992Z","iopub.execute_input":"2025-11-30T23:35:51.690403Z","iopub.status.idle":"2025-11-30T23:42:30.488618Z","shell.execute_reply.started":"2025-11-30T23:35:51.690362Z","shell.execute_reply":"2025-11-30T23:42:30.487826Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 6: Prepare Training Data\nfeature_cols = [col for col in features.columns if col not in [\n    'object_id', 'target', 'target_original', 'hostgal_specz', 'ra', 'decl'\n]]\n\nX = features[feature_cols].values\ny = features['target'].values\n\nprint(f\"\\nTraining data shape: X={X.shape}, y={y.shape}\")\nprint(f\"Number of features: {len(feature_cols)}\")\n\nclasses = np.unique(y)\nclass_weights = compute_class_weight('balanced', classes=classes, y=y)\nclass_weight_dict = dict(zip(classes, class_weights))\n\nprint(\"\\nClass distribution:\")\nprint(pd.Series(y).value_counts().sort_index())","metadata":{"_uuid":"69a98251-4c12-455e-81ea-ee0abde669f6","_cell_guid":"9f4c4f4a-aeca-4e63-945d-d8c80736fcc4","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:42:30.489398Z","iopub.execute_input":"2025-11-30T23:42:30.489643Z","iopub.status.idle":"2025-11-30T23:42:30.511429Z","shell.execute_reply.started":"2025-11-30T23:42:30.489623Z","shell.execute_reply":"2025-11-30T23:42:30.510834Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 7: LightGBM Training with Cross-Validation\nprint(\"\\nLightGBM Training with Weighted Log Loss\")\n\nclass_weights_plasticc = {6: 1, 15: 1, 16: 1, 42: 1, 52: 1, 53: 1, 62: 1,\n                          64: 1, 65: 1, 67: 1, 88: 1, 90: 1, 92: 1, 95: 1, 99: 2}\n\nmapped_weights = {}\nfor c, w in class_weights_plasticc.items():\n    if c in class_mapping:\n        mapped_weights[class_mapping[c]] = w\n    else:\n        mapped_weights[c] = 1\n\nparams = {\n    'boosting_type': 'gbdt',\n    'objective': 'multiclass',\n    'metric': 'multi_logloss',\n    'num_class': len(class_mapping),\n    'learning_rate': 0.01,\n    'feature_fraction': 0.8,\n    'bagging_fraction': 0.8,\n    'bagging_freq': 3,\n    'num_leaves': 300,\n    'max_depth': 15,\n    'min_data_in_leaf': 5,\n    'lambda_l1': 0.5,\n    'lambda_l2': 0.5,\n    'min_gain_to_split': 0.05,\n    'feature_fraction_bynode': 0.8,\n    'verbose': -1,\n    'seed': 42,\n    'num_threads': -1,\n    'extra_trees': True,\n    'path_smooth': 0.1,\n    'max_bin': 255,\n    'min_sum_hessian_in_leaf': 0.1,\n    'subsample_for_bin': 200000,\n}\n\nn_folds = 5\nskf = StratifiedKFold(n_splits=n_folds, shuffle=True, random_state=42)\n\ncv_scores = []\nfeature_importance_list = []\npredictions = np.zeros((len(X), params['num_class']))\nmodels = []\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"\\nFold {fold + 1}/{n_folds}\")\n    \n    X_train, X_val = X[train_idx], X[val_idx]\n    y_train, y_val = y[train_idx], y[val_idx]\n    \n    sample_weights = np.array([mapped_weights.get(label, 1) for label in y_train])\n    val_weights = np.array([mapped_weights.get(label, 1) for label in y_val])\n    \n    train_data = lgb.Dataset(X_train, label=y_train, weight=sample_weights)\n    val_data = lgb.Dataset(X_val, label=y_val, weight=val_weights, reference=train_data)\n    \n    model = lgb.train(\n        params,\n        train_data,\n        valid_sets=[val_data],\n        num_boost_round=2000,\n        callbacks=[\n            lgb.early_stopping(200),\n            lgb.log_evaluation(200)\n        ]\n    )\n    \n    models.append(model)\n    \n    val_preds = model.predict(X_val, num_iteration=model.best_iteration)\n    predictions[val_idx] = val_preds\n    \n    val_score = 0\n    for i in range(len(y_val)):\n        prob = val_preds[i][y_val[i]]\n        val_score += -mapped_weights.get(y_val[i], 1) * np.log(max(prob, 1e-15))\n    val_score /= sum(val_weights)\n    cv_scores.append(val_score)\n    \n    print(f\"Fold {fold + 1} Weighted Log Loss: {val_score:.4f}\")\n    \n    fold_accuracy = (y_val == np.argmax(val_preds, axis=1)).mean()\n    print(f\"Fold {fold + 1} Accuracy: {fold_accuracy:.4f}\")\n    \n    importance = model.feature_importance(importance_type='gain')\n    for i, feat in enumerate(feature_cols):\n        feature_importance_list.append({\n            'feature': feat,\n            'importance': importance[i],\n            'fold': fold + 1\n        })\n\nfeature_importance = pd.DataFrame(feature_importance_list)\n\nprint(f\"\\nLightGBM Mean Weighted Log Loss: {np.mean(cv_scores):.4f} (+/- {np.std(cv_scores):.4f})\")\nprint(f\"Overall Accuracy: {(y == np.argmax(predictions, axis=1)).mean():.4f}\")","metadata":{"_uuid":"72f477ed-9e92-43d3-ad8d-0a83ecfe4b81","_cell_guid":"b095d89a-dd03-46da-9630-474751fdd0cf","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:42:30.512170Z","iopub.execute_input":"2025-11-30T23:42:30.512409Z","iopub.status.idle":"2025-11-30T23:47:27.578783Z","shell.execute_reply.started":"2025-11-30T23:42:30.512388Z","shell.execute_reply":"2025-11-30T23:47:27.578187Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 8: Feature Importance Analysis\nprint(\"\\nFeature Importance Analysis\")\n\nimportance_summary = feature_importance.groupby('feature').agg({\n    'importance': ['mean', 'std', 'count']\n}).round(2)\n\nimportance_summary.columns = ['mean_importance', 'std_importance', 'count']\nimportance_summary = importance_summary.sort_values('mean_importance', ascending=False)\n\ntop_20_features = importance_summary.head(20)\n\n# IEEE ডাবল কলাম - অনেক বড় ফন্ট\nplt.rcParams.update({\n    'font.family': 'serif',\n    'font.serif': ['Times New Roman', 'Times', 'DejaVu Serif'],\n    'mathtext.fontset': 'stix',\n    'font.size': 20,\n    'axes.titlesize': 26,\n    'axes.labelsize': 24,\n    'xtick.labelsize': 20,\n    'ytick.labelsize': 18,\n    'axes.linewidth': 1.5,\n    'lines.linewidth': 2.0,\n})\n\nplt.figure(figsize=(16, 14))\ny_pos = np.arange(len(top_20_features))\n\nbars = plt.barh(y_pos, top_20_features['mean_importance'],\n                xerr=top_20_features['std_importance'],\n                align='center', alpha=0.85, capsize=8,\n                color='#2196F3', edgecolor='black', linewidth=1.5,\n                error_kw={'elinewidth': 3, 'capthick': 3})\n\nplt.yticks(y_pos, top_20_features.index, fontsize=18)\nplt.xticks(fontsize=20)\nplt.xlabel('Feature Importance (Gain)', fontsize=24, fontweight='bold', labelpad=15)\nplt.title(f'Top 20 Feature Importances\\nAveraged across {n_folds} folds', \n          fontsize=26, fontweight='bold', pad=20)\nplt.gca().invert_yaxis()\nplt.grid(axis='x', alpha=0.4, linewidth=1.5)\n\n# X-axis এর রেঞ্জ বাড়ানো যাতে টেক্সট ফিট হয়\nmax_val = top_20_features['mean_importance'].max() + top_20_features['std_importance'].max()\nplt.xlim(0, max_val * 1.30)\n\n# প্রতিটি বারে ভ্যালু দেখানো - অনেক বড় ফন্ট\nfor i, (bar, val, std) in enumerate(zip(bars, top_20_features['mean_importance'], \n                                         top_20_features['std_importance'])):\n    plt.text(val + std + max_val * 0.04, \n             bar.get_y() + bar.get_height()/2,\n             f'{val:.1f}', va='center', \n             fontsize=20, fontweight='bold', color='black')\n\nplt.tick_params(axis='both', width=2, length=8)\nplt.tight_layout()\n# আগে ছিল:\n# plt.savefig('feature_importance.png', dpi=600, bbox_inches='tight')\n\n# এখন এটা করো:\nplt.savefig('feature_importance.pdf', format='pdf', bbox_inches='tight')\nplt.savefig('feature_importance.eps', format='eps', bbox_inches='tight')\nplt.savefig('feature_importance.png', dpi=600, bbox_inches='tight')  # backup PNG ও রাখো\nplt.show()\n\nmean_importance = importance_summary['mean_importance']\n\nprint(\"\\nTop 10 Most Important Features:\")\nfor i, (feat, row) in enumerate(top_20_features.head(10).iterrows(), 1):\n    print(f\"{i:2d}. {feat:30s}: {row['mean_importance']:6.2f} (±{row['std_importance']:5.2f})\")","metadata":{"_uuid":"2bb4ade9-92c0-4e53-9421-56fed21efb64","_cell_guid":"29fd6a65-a180-46ec-8cfd-dfdfa3da8663","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:47:27.579485Z","iopub.execute_input":"2025-11-30T23:47:27.579702Z","iopub.status.idle":"2025-11-30T23:47:28.482653Z","shell.execute_reply.started":"2025-11-30T23:47:27.579683Z","shell.execute_reply":"2025-11-30T23:47:28.481977Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 9: Confusion Matrix\nprint(\"\\nConfusion Matrix\")\n\ny_pred = np.argmax(predictions, axis=1)\n\ny_original = features['target_original'].values\ny_pred_original = [reverse_mapping[pred] for pred in y_pred]\n\ncm = confusion_matrix(y_original, y_pred_original, labels=sorted(train_meta['target_original'].unique()))\n\n# IEEE ডাবল কলাম - অনেক বড় ফন্ট\nplt.rcParams.update({\n    'font.family': 'serif',\n    'font.serif': ['Times New Roman', 'Times', 'DejaVu Serif'],\n    'mathtext.fontset': 'stix',\n    'font.size': 20,\n    'axes.titlesize': 26,\n    'axes.labelsize': 24,\n    'xtick.labelsize': 18,\n    'ytick.labelsize': 18,\n    'axes.linewidth': 1.5,\n    'lines.linewidth': 2.0,\n})\n\nplt.figure(figsize=(18, 16))\nsns.heatmap(cm, annot=True, fmt='d', cmap='Blues',\n            xticklabels=sorted(train_meta['target_original'].unique()),\n            yticklabels=sorted(train_meta['target_original'].unique()),\n            annot_kws={'size': 22, 'fontweight': 'bold'},\n            linewidths=1.5,\n            linecolor='white',\n            cbar_kws={'shrink': 0.8})\n\nplt.xlabel('Predicted Class', fontsize=24, fontweight='bold', labelpad=20)\nplt.ylabel('Actual Class', fontsize=24, fontweight='bold', labelpad=20)\nplt.title('Confusion Matrix - PLAsTiCC Classification', fontsize=26, fontweight='bold', pad=25)\nplt.xticks(fontsize=18, rotation=45, ha='right')\nplt.yticks(fontsize=18, rotation=0)\n\n# Colorbar ফন্ট বড় করা\ncbar = plt.gca().collections[0].colorbar\ncbar.ax.tick_params(labelsize=18)\n\nplt.tight_layout()\n# আগে ছিল:\n# plt.savefig('confusion_matrix.png', dpi=600, bbox_inches='tight')\n\n# এখন এটা করো:\nplt.savefig('confusion_matrix.pdf', format='pdf', bbox_inches='tight')\nplt.savefig('confusion_matrix.eps', format='eps', bbox_inches='tight')\nplt.savefig('confusion_matrix.png', dpi=600, bbox_inches='tight')  # backup\nplt.show()\n\nprint(\"\\nClassification Report:\")\nprint(classification_report(y_original, y_pred_original,\n                          target_names=[str(c) for c in sorted(train_meta['target_original'].unique())]))","metadata":{"_uuid":"f957200c-cd4e-44fe-9254-109b9efb3155","_cell_guid":"c0189200-ef6b-4fb5-92f4-b196d2782708","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:47:28.483548Z","iopub.execute_input":"2025-11-30T23:47:28.484074Z","iopub.status.idle":"2025-11-30T23:47:30.011327Z","shell.execute_reply.started":"2025-11-30T23:47:28.484047Z","shell.execute_reply":"2025-11-30T23:47:30.010732Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 10: LightGBM Results Summary\nprint(\"\\nLightGBM Results Summary\")\n\nlgb_results = pd.DataFrame({\n    'Metric': ['Log Loss', 'Accuracy', 'Macro F1', 'Weighted F1'],\n    'Score': [\n        np.mean(cv_scores),\n        (y == y_pred).mean(),\n        classification_report(y, y_pred, output_dict=True)['macro avg']['f1-score'],\n        classification_report(y, y_pred, output_dict=True)['weighted avg']['f1-score']\n    ]\n})\n\nprint(lgb_results.to_string(index=False))","metadata":{"_uuid":"1a17f944-e8da-4f23-9416-193e461276f5","_cell_guid":"9ee0a24f-cb6c-4532-9e6f-1c71e821c768","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:47:30.012116Z","iopub.execute_input":"2025-11-30T23:47:30.012334Z","iopub.status.idle":"2025-11-30T23:47:30.048927Z","shell.execute_reply.started":"2025-11-30T23:47:30.012308Z","shell.execute_reply":"2025-11-30T23:47:30.048210Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 11: Multiple Algorithms Comparison\nprint(\"\\nMultiple Algorithms Comparison\")\n\nn_classes = len(np.unique(y))\nresults_comparison = []\n\nprint(\"\\n1. Training Random Forest...\")\nstart_time = time.time()\n\nrf_scores = []\nrf_predictions = np.zeros((len(X), n_classes))\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"RF Fold {fold + 1}/{n_folds}\", end=' ')\n    \n    rf = RandomForestClassifier(\n        n_estimators=1000,\n        max_depth=30,\n        min_samples_split=3,\n        min_samples_leaf=1,\n        max_features='sqrt',\n        random_state=42,\n        n_jobs=-1,\n        class_weight='balanced_subsample',\n        criterion='entropy',\n        bootstrap=True,\n        oob_score=True\n    )\n    \n    rf.fit(X[train_idx], y[train_idx])\n    \n    rf_pred_raw = rf.predict_proba(X[val_idx])\n    \n    if rf_pred_raw.shape[1] < n_classes:\n        rf_pred = np.zeros((len(val_idx), n_classes))\n        for i, class_label in enumerate(rf.classes_):\n            rf_pred[:, class_label] = rf_pred_raw[:, i]\n    else:\n        rf_pred = rf_pred_raw\n    \n    rf_predictions[val_idx] = rf_pred\n    \n    scores = []\n    val_weights = np.array([mapped_weights.get(label, 1) for label in y[val_idx]])\n    for i in range(len(val_idx)):\n        prob = rf_pred[i, y[val_idx][i]]\n        scores.append(-mapped_weights.get(y[val_idx][i], 1) * np.log(max(prob, 1e-15)))\n    score = np.sum(scores) / np.sum(val_weights)\n    rf_scores.append(score)\n    print(f\"Weighted Loss: {score:.4f}\")\n\nrf_time = time.time() - start_time\nprint(f\"\\nRF Mean Weighted Log Loss: {np.mean(rf_scores):.4f} (Time: {rf_time:.1f}s)\")\nprint(f\"RF Overall Accuracy: {(y == np.argmax(rf_predictions, axis=1)).mean():.4f}\")\n\nresults_comparison.append({\n    'Algorithm': 'Random Forest',\n    'Log Loss': np.mean(rf_scores),\n    'Std': np.std(rf_scores),\n    'Accuracy': (y == np.argmax(rf_predictions, axis=1)).mean(),\n    'Time (s)': rf_time\n})\n\nprint(\"\\n2. Training XGBoost...\")\ntry:\n    from xgboost import XGBClassifier\n    start_time = time.time()\n    \n    xgb_scores = []\n    xgb_predictions = np.zeros((len(X), n_classes))\n    successful_folds = 0\n    \n    for fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n        print(f\"XGB Fold {fold + 1}/{n_folds}\", end=' ')\n        \n        try:\n            xgb = XGBClassifier(\n                n_estimators=1000,\n                max_depth=15,\n                learning_rate=0.01,\n                subsample=0.8,\n                colsample_bytree=0.8,\n                colsample_bylevel=0.8,\n                colsample_bynode=0.8,\n                reg_alpha=0.5,\n                reg_lambda=1.0,\n                min_child_weight=1,\n                gamma=0.05,\n                random_state=42,\n                use_label_encoder=False,\n                eval_metric='mlogloss',\n                tree_method='hist'\n            )\n            \n            sample_weights = np.array([mapped_weights.get(y[idx], 1) for idx in train_idx])\n            \n            xgb.fit(\n                X[train_idx], y[train_idx],\n                sample_weight=sample_weights,\n                eval_set=[(X[val_idx], y[val_idx])],\n                early_stopping_rounds=150,\n                verbose=False\n            )\n            \n            xgb_pred = xgb.predict_proba(X[val_idx])\n            xgb_predictions[val_idx] = xgb_pred\n            \n            scores = []\n            val_weights = np.array([mapped_weights.get(label, 1) for label in y[val_idx]])\n            for i in range(len(val_idx)):\n                prob = xgb_pred[i, y[val_idx][i]]\n                scores.append(-mapped_weights.get(y[val_idx][i], 1) * np.log(max(prob, 1e-15)))\n            score = np.sum(scores) / np.sum(val_weights)\n            xgb_scores.append(score)\n            successful_folds += 1\n            print(f\"Weighted Loss: {score:.4f}\")\n                \n        except Exception as e:\n            print(f\"Error: {str(e)[:50]}...\")\n            continue\n    \n    if successful_folds > 0:\n        xgb_time = time.time() - start_time\n        print(f\"\\nXGB Mean Weighted Log Loss: {np.mean(xgb_scores):.4f} (Time: {xgb_time:.1f}s)\")\n        print(f\"XGB Overall Accuracy: {(y == np.argmax(xgb_predictions, axis=1)).mean():.4f}\")\n        \n        results_comparison.append({\n            'Algorithm': 'XGBoost',\n            'Log Loss': np.mean(xgb_scores),\n            'Std': np.std(xgb_scores) if len(xgb_scores) > 1 else 0,\n            'Accuracy': (y == np.argmax(xgb_predictions, axis=1)).mean(),\n            'Time (s)': xgb_time\n        })\n    else:\n        print(\"\\nXGBoost failed on all folds\")\n        \nexcept ImportError:\n    print(\"XGBoost not installed\")\n\nprint(\"\\n3. Training Extra Trees...\")\nstart_time = time.time()\n\net_scores = []\net_predictions = np.zeros((len(X), n_classes))\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"ET Fold {fold + 1}/{n_folds}\", end=' ')\n    \n    et = ExtraTreesClassifier(\n        n_estimators=1000,\n        max_depth=30,\n        min_samples_split=3,\n        min_samples_leaf=1,\n        random_state=42,\n        n_jobs=-1,\n        class_weight='balanced_subsample',\n        criterion='entropy',\n        bootstrap=False,\n        max_features='sqrt'\n    )\n    \n    et.fit(X[train_idx], y[train_idx])\n    \n    et_pred_raw = et.predict_proba(X[val_idx])\n    \n    if et_pred_raw.shape[1] < n_classes:\n        et_pred = np.zeros((len(val_idx), n_classes))\n        for i, class_label in enumerate(et.classes_):\n            et_pred[:, class_label] = et_pred_raw[:, i]\n    else:\n        et_pred = et_pred_raw\n    \n    et_predictions[val_idx] = et_pred\n    \n    scores = []\n    val_weights = np.array([mapped_weights.get(label, 1) for label in y[val_idx]])\n    for i in range(len(val_idx)):\n        prob = et_pred[i, y[val_idx][i]]\n        scores.append(-mapped_weights.get(y[val_idx][i], 1) * np.log(max(prob, 1e-15)))\n    score = np.sum(scores) / np.sum(val_weights)\n    et_scores.append(score)\n    print(f\"Weighted Loss: {score:.4f}\")\n\net_time = time.time() - start_time\nprint(f\"\\nET Mean Weighted Log Loss: {np.mean(et_scores):.4f} (Time: {et_time:.1f}s)\")\nprint(f\"ET Overall Accuracy: {(y == np.argmax(et_predictions, axis=1)).mean():.4f}\")\n\nresults_comparison.append({\n    'Algorithm': 'Extra Trees',\n    'Log Loss': np.mean(et_scores),\n    'Std': np.std(et_scores),\n    'Accuracy': (y == np.argmax(et_predictions, axis=1)).mean(),\n    'Time (s)': et_time\n})\n\nresults_comparison.append({\n    'Algorithm': 'LightGBM',\n    'Log Loss': np.mean(cv_scores),\n    'Std': np.std(cv_scores),\n    'Accuracy': (y == np.argmax(predictions, axis=1)).mean(),\n    'Time (s)': 0\n})\n\nprint(\"\\nAlgorithms trained successfully\")\n\nif len(results_comparison) > 0:\n    results_df = pd.DataFrame(results_comparison)\n    print(\"\\nResults Summary:\")\n    print(results_df.to_string(index=False))","metadata":{"_uuid":"1cdc9ad8-fd3b-4f70-9608-25c61e886718","_cell_guid":"bb19abf4-3830-4c7d-9b92-9592d4ed71d8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2025-11-30T23:47:30.051045Z","iopub.execute_input":"2025-11-30T23:47:30.051451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 12: Neural Network Training\nprint(\"\\n4. Training Neural Network...\")\nstart_time = time.time()\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\nmlp_scores = []\nmlp_predictions = np.zeros((len(X), len(class_mapping)))\n\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X_scaled, y)):\n    print(f\"MLP Fold {fold + 1}/{n_folds}\", end=' ')\n    \n    mlp = MLPClassifier(\n        hidden_layer_sizes=(100, 50),\n        max_iter=100,\n        random_state=42,\n        early_stopping=True,\n        validation_fraction=0.2,\n        verbose=False,\n        alpha=0.001,\n        learning_rate_init=0.001\n    )\n    \n    mlp.fit(X_scaled[train_idx], y[train_idx])\n    \n    mlp_pred = mlp.predict_proba(X_scaled[val_idx])\n    \n    if mlp_pred.shape[1] < len(class_mapping):\n        mlp_pred_full = np.zeros((len(val_idx), len(class_mapping)))\n        for i, class_label in enumerate(mlp.classes_):\n            mlp_pred_full[:, class_label] = mlp_pred[:, i]\n        mlp_pred = mlp_pred_full\n    \n    mlp_predictions[val_idx] = mlp_pred\n    \n    scores = []\n    val_weights = np.array([mapped_weights.get(label, 1) for label in y[val_idx]])\n    for i in range(len(val_idx)):\n        prob = mlp_pred[i, y[val_idx][i]]\n        scores.append(-mapped_weights.get(y[val_idx][i], 1) * np.log(max(prob, 1e-15)))\n    score = np.sum(scores) / np.sum(val_weights)\n    mlp_scores.append(score)\n    print(f\"Weighted Loss: {score:.4f}\")\n\nmlp_time = time.time() - start_time\nprint(f\"\\nMLP Mean Weighted Log Loss: {np.mean(mlp_scores):.4f} (Time: {mlp_time:.1f}s)\")\nprint(f\"MLP Overall Accuracy: {(y == np.argmax(mlp_predictions, axis=1)).mean():.4f}\")\n\nresults_comparison.append({\n    'Algorithm': 'Neural Network',\n    'Log Loss': np.mean(mlp_scores),\n    'Std': np.std(mlp_scores),\n    'Accuracy': (y == np.argmax(mlp_predictions, axis=1)).mean(),\n    'Time (s)': mlp_time\n})\n\nprint(\"\\nNeural Network training completed\")","metadata":{"_uuid":"34869bc2-f747-49f7-a859-eebed35c5bd6","_cell_guid":"dd1e840b-273c-48f6-af86-e2cb7173b38b","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 13: Results Comparison Visualization\nresults_df = pd.DataFrame(results_comparison).sort_values('Log Loss')\nprint(\"\\nAlgorithm Comparison Results:\")\nprint(results_df.to_string(index=False))\n\nbest_base_model = results_df.iloc[0]\nfor idx, row in results_df.iterrows():\n    improvement = (best_base_model['Log Loss'] - row['Log Loss']) / best_base_model['Log Loss'] * 100\n    print(f\"{row['Algorithm']:20s}: {improvement:+6.2f}% vs best model\")\n\nresults_df.to_csv('algorithm_comparison_final.csv', index=False)\n\n# IEEE ডাবল কলাম - অনেক বড় ফন্ট\nplt.rcParams.update({\n    'font.family': 'serif',\n    'font.serif': ['Times New Roman', 'Times', 'DejaVu Serif'],\n    'mathtext.fontset': 'stix',\n    'font.size': 20,\n    'axes.titlesize': 24,\n    'axes.labelsize': 22,\n    'xtick.labelsize': 18,\n    'ytick.labelsize': 18,\n    'legend.fontsize': 18,\n    'figure.titlesize': 26,\n    'axes.linewidth': 1.5,\n    'lines.linewidth': 2.0,\n})\n\nfig, (ax1, ax2) = plt.subplots(1, 2, figsize=(20, 9))\n\nalgorithms = results_df['Algorithm']\nlog_losses = results_df['Log Loss']\nerrors = results_df['Std']\n\ncolors = ['#1f77b4', '#ff7f0e', '#2ca02c', '#d62728', '#9467bd']\nbars1 = ax1.bar(algorithms, log_losses, yerr=errors, capsize=8, \n                color=colors[:len(algorithms)], edgecolor='black', linewidth=2)\n\nax1.set_xlabel('Algorithm', fontsize=22, fontweight='bold', labelpad=15)\nax1.set_ylabel('Log Loss', fontsize=22, fontweight='bold', labelpad=15)\nax1.set_title('Log Loss Comparison', fontsize=24, fontweight='bold', pad=20)\nax1.set_xticklabels(algorithms, rotation=30, ha='right', fontsize=18)\nax1.tick_params(axis='y', labelsize=18)\nax1.tick_params(axis='both', width=2, length=8)\nax1.grid(axis='y', alpha=0.4, linewidth=1.5)\n\n# বারের উপরে টেক্সট - অনেক বড় ফন্ট\nmax_height1 = max(log_losses) + max(errors)\nax1.set_ylim(0, max_height1 * 1.22)\nfor bar, loss, err in zip(bars1, log_losses, errors):\n    ax1.text(bar.get_x() + bar.get_width()/2, bar.get_height() + err + 0.025,\n             f'{loss:.3f}', ha='center', va='bottom', \n             fontsize=22, fontweight='bold', color='black')\n\naccuracies = results_df['Accuracy']\nbars2 = ax2.bar(algorithms, accuracies, color=colors[:len(algorithms)], \n                edgecolor='black', linewidth=2)\n\nax2.set_xlabel('Algorithm', fontsize=22, fontweight='bold', labelpad=15)\nax2.set_ylabel('Accuracy', fontsize=22, fontweight='bold', labelpad=15)\nax2.set_title('Accuracy Comparison', fontsize=24, fontweight='bold', pad=20)\nax2.set_xticklabels(algorithms, rotation=30, ha='right', fontsize=18)\nax2.tick_params(axis='y', labelsize=18)\nax2.tick_params(axis='both', width=2, length=8)\nax2.grid(axis='y', alpha=0.4, linewidth=1.5)\n\n# বারের উপরে টেক্সট - অনেক বড় ফন্ট\nmax_height2 = max(accuracies)\nax2.set_ylim(0, max_height2 * 1.15)\nfor bar, acc in zip(bars2, accuracies):\n    ax2.text(bar.get_x() + bar.get_width()/2, bar.get_height() + 0.015,\n             f'{acc:.3f}', ha='center', va='bottom', \n             fontsize=22, fontweight='bold', color='black')\n\nplt.suptitle('Algorithm Performance Comparison on PLAsTiCC Dataset', \n             fontsize=26, fontweight='bold', y=1.02)\nplt.tight_layout()\nplt.subplots_adjust(top=0.88, bottom=0.18, wspace=0.3)\nplt.savefig('algorithm_comparison.png', dpi=600, bbox_inches='tight')\n# আগে ছিল:\n# plt.savefig('algorithm_comparison.png', dpi=600, bbox_inches='tight')\n\n# এখন এটা করো:\nplt.savefig('algorithm_comparison.pdf', format='pdf', bbox_inches='tight')\nplt.savefig('algorithm_comparison.eps', format='eps', bbox_inches='tight')\nplt.savefig('algorithm_comparison.png', dpi=600, bbox_inches='tight')  # backup\nplt.show()","metadata":{"_uuid":"084eb836-e2c7-4d8a-9e13-6e138d341a0b","_cell_guid":"b22a9147-73a4-4d43-8223-04c067bbe3a6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 14: Ensemble Model\nprint(\"\\nAdvanced Ensemble Model\")\n\nclass LGBWrapper(BaseEstimator, ClassifierMixin):\n    def __init__(self, params, num_rounds=2000):\n        self.params = params\n        self.num_rounds = num_rounds\n        self.model = None\n        \n    def fit(self, X, y, sample_weight=None):\n        train_data = lgb.Dataset(X, label=y, weight=sample_weight)\n        self.model = lgb.train(\n            self.params,\n            train_data,\n            num_boost_round=self.num_rounds,\n            callbacks=[lgb.log_evaluation(0)]\n        )\n        return self\n    \n    def predict_proba(self, X):\n        return self.model.predict(X, num_iteration=self.model.best_iteration)\n    \n    def predict(self, X):\n        return np.argmax(self.predict_proba(X), axis=1)\n\nlgb_model = LGBWrapper(params)\n\nrf_model = RandomForestClassifier(\n    n_estimators=1000,\n    max_depth=30,\n    min_samples_split=3,\n    min_samples_leaf=1,\n    class_weight='balanced_subsample',\n    criterion='entropy',\n    random_state=42,\n    n_jobs=-1\n)\n\ntry:\n    from xgboost import XGBClassifier\n    xgb_model = XGBClassifier(\n        n_estimators=1000,\n        max_depth=15,\n        learning_rate=0.01,\n        subsample=0.8,\n        colsample_bytree=0.8,\n        random_state=42,\n        use_label_encoder=False,\n        eval_metric='mlogloss'\n    )\n    use_xgb = True\nexcept:\n    use_xgb = False\n\net_model = ExtraTreesClassifier(\n    n_estimators=1000,\n    max_depth=30,\n    min_samples_split=3,\n    min_samples_leaf=1,\n    class_weight='balanced_subsample',\n    random_state=42,\n    n_jobs=-1\n)\n\nif use_xgb:\n    ensemble = VotingClassifier(\n        estimators=[\n            ('lgb', lgb_model),\n            ('rf', rf_model),\n            ('xgb', xgb_model),\n            ('et', et_model)\n        ],\n        voting='soft',\n        weights=[0.4, 0.3, 0.2, 0.1]\n    )\nelse:\n    ensemble = VotingClassifier(\n        estimators=[\n            ('lgb', lgb_model),\n            ('rf', rf_model),\n            ('et', et_model)\n        ],\n        voting='soft',\n        weights=[0.5, 0.3, 0.2]\n    )\n\nensemble_scores = []\nensemble_predictions = np.zeros((len(X), len(class_mapping)))\n\nprint(\"Training ensemble with cross-validation...\")\nfor fold, (train_idx, val_idx) in enumerate(skf.split(X, y)):\n    print(f\"\\nEnsemble Fold {fold + 1}/{n_folds}\")\n    \n    ensemble.fit(X[train_idx], y[train_idx])\n    \n    ensemble_pred = ensemble.predict_proba(X[val_idx])\n    ensemble_predictions[val_idx] = ensemble_pred\n    \n    val_score = 0\n    val_weights = np.array([mapped_weights.get(label, 1) for label in y[val_idx]])\n    for i in range(len(val_idx)):\n        prob = ensemble_pred[i, y[val_idx][i]]\n        val_score += -mapped_weights.get(y[val_idx][i], 1) * np.log(max(prob, 1e-15))\n    val_score /= sum(val_weights)\n    ensemble_scores.append(val_score)\n    \n    fold_accuracy = (y[val_idx] == np.argmax(ensemble_pred, axis=1)).mean()\n    print(f\"Fold {fold + 1} Weighted Log Loss: {val_score:.4f}\")\n    print(f\"Fold {fold + 1} Accuracy: {fold_accuracy:.4f}\")\n\nensemble_mean_loss = np.mean(ensemble_scores)\nensemble_accuracy = (y == np.argmax(ensemble_predictions, axis=1)).mean()\n\nprint(f\"\\nEnsemble Mean Weighted Log Loss: {ensemble_mean_loss:.4f} (+/- {np.std(ensemble_scores):.4f})\")\nprint(f\"Ensemble Overall Accuracy: {ensemble_accuracy:.4f}\")\n\nbest_single = min(results_comparison, key=lambda x: x['Log Loss'])\nprint(f\"\\nBest Single Model: {best_single['Algorithm']}\")\nprint(f\"Best Single Log Loss: {best_single['Log Loss']:.4f}\")\nprint(f\"Best Single Accuracy: {best_single['Accuracy']:.4f}\")\n\nprint(f\"\\nEnsemble Improvement:\")\nprint(f\"Log Loss: {(best_single['Log Loss'] - ensemble_mean_loss)/best_single['Log Loss']*100:+.2f}%\")\nprint(f\"Accuracy: {(ensemble_accuracy - best_single['Accuracy'])*100:+.2f}%\")\n\nprint(\"\\nAdvanced ensemble training completed\")","metadata":{"_uuid":"2ab9830c-867f-4a4b-a7c0-4e9846294017","_cell_guid":"426ed4e0-16c6-46ff-8aff-c0a6fe0cf6b8","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 15: Test Set Evaluation\nprint(\"\\nTest Set Evaluation\")\n\nX_train, X_test, y_train, y_test = train_test_split(\n    X, y, test_size=0.2, stratify=y, random_state=42\n)\n\nprint(f\"Train set size: {X_train.shape[0]} objects\")\nprint(f\"Test set size: {X_test.shape[0]} objects\")\n\nprint(\"\\nTraining models on train set...\")\n\nprint(\"Training LightGBM...\")\ntrain_data = lgb.Dataset(\n    X_train, label=y_train,\n    weight=[mapped_weights.get(label, 1) for label in y_train]\n)\n\nlgb_model = lgb.train(\n    params,\n    train_data,\n    num_boost_round=1000,\n    callbacks=[lgb.log_evaluation(0)]\n)\nlgb_pred_test = lgb_model.predict(X_test, num_iteration=lgb_model.best_iteration)\nlgb_pred_train = lgb_model.predict(X_train, num_iteration=lgb_model.best_iteration)\n\nprint(\"Training Random Forest...\")\nrf_model = RandomForestClassifier(\n    n_estimators=500, max_depth=25, min_samples_split=5,\n    min_samples_leaf=2, random_state=42, n_jobs=-1,\n    class_weight='balanced'\n)\nrf_model.fit(X_train, y_train)\nrf_pred_test = rf_model.predict_proba(X_test)\nrf_pred_train = rf_model.predict_proba(X_train)\n\nprint(\"Training Extra Trees...\")\net_model = ExtraTreesClassifier(\n    n_estimators=500, max_depth=25, min_samples_split=5,\n    min_samples_leaf=2, random_state=42, n_jobs=-1,\n    class_weight='balanced'\n)\net_model.fit(X_train, y_train)\net_pred_test = et_model.predict_proba(X_test)\net_pred_train = et_model.predict_proba(X_train)\n\ntry:\n    from xgboost import XGBClassifier\n    print(\"Training XGBoost...\")\n    xgb_model = XGBClassifier(\n        n_estimators=500, max_depth=12, learning_rate=0.02,\n        subsample=0.8, colsample_bytree=0.8, random_state=42,\n        use_label_encoder=False, eval_metric='mlogloss'\n    )\n    xgb_model.fit(\n        X_train, y_train,\n        sample_weight=[mapped_weights.get(label, 1) for label in y_train]\n    )\n    xgb_pred_test = xgb_model.predict_proba(X_test)\n    xgb_pred_train = xgb_model.predict_proba(X_train)\n    use_xgb = True\nexcept:\n    print(\"XGBoost not available, skipping...\")\n    xgb_pred_test = None\n    xgb_pred_train = None\n    use_xgb = False\n\nprint(\"Training Neural Network...\")\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\nmlp_model = MLPClassifier(\n    hidden_layer_sizes=(100, 50), max_iter=100, random_state=42,\n    early_stopping=True, validation_fraction=0.2\n)\nmlp_model.fit(X_train_scaled, y_train)\nmlp_pred_test = mlp_model.predict_proba(X_test_scaled)\nmlp_pred_train = mlp_model.predict_proba(X_train_scaled)\n\nprint(\"\\nCreating ensemble predictions...\")\n\nif use_xgb:\n    ensemble_weights = {\n        'LightGBM': 0.35,\n        'RandomForest': 0.20,\n        'ExtraTrees': 0.15,\n        'XGBoost': 0.25,\n        'NeuralNetwork': 0.05\n    }\n    predictions_test = [lgb_pred_test, rf_pred_test, et_pred_test, xgb_pred_test, mlp_pred_test]\n    predictions_train = [lgb_pred_train, rf_pred_train, et_pred_train, xgb_pred_train, mlp_pred_train]\n    model_names = ['LightGBM', 'RandomForest', 'ExtraTrees', 'XGBoost', 'NeuralNetwork']\nelse:\n    ensemble_weights = {\n        'LightGBM': 0.40,\n        'RandomForest': 0.30,\n        'ExtraTrees': 0.20,\n        'NeuralNetwork': 0.10\n    }\n    predictions_test = [lgb_pred_test, rf_pred_test, et_pred_test, mlp_pred_test]\n    predictions_train = [lgb_pred_train, rf_pred_train, et_pred_train, mlp_pred_train]\n    model_names = ['LightGBM', 'RandomForest', 'ExtraTrees', 'NeuralNetwork']\n\nensemble_pred_test = np.zeros_like(lgb_pred_test)\nensemble_pred_train = np.zeros_like(lgb_pred_train)\n\nfor name, pred_test, pred_train in zip(model_names, predictions_test, predictions_train):\n    weight = ensemble_weights[name]\n    ensemble_pred_test += weight * pred_test\n    ensemble_pred_train += weight * pred_train\n\nprint(\"\\nIndividual Model Performance on Test Set:\")\n\nfor name, pred_test in zip(model_names, predictions_test):\n    test_loss = 0\n    test_weights_sum = 0\n    for i in range(len(y_test)):\n        prob = pred_test[i, y_test[i]]\n        weight = mapped_weights.get(y_test[i], 1)\n        test_loss += -weight * np.log(max(prob, 1e-15))\n        test_weights_sum += weight\n    test_loss /= test_weights_sum\n    \n    test_accuracy = (y_test == np.argmax(pred_test, axis=1)).mean()\n    \n    print(f\"{name:15s}: Loss = {test_loss:.4f}, Accuracy = {test_accuracy:.4f}\")\n\nprint(\"\\nEnsemble Performance on Test Set:\")\n\ntest_weighted_loss = 0\ntest_weights = np.array([mapped_weights.get(label, 1) for label in y_test])\n\nfor i in range(len(y_test)):\n    prob = ensemble_pred_test[i, y_test[i]]\n    weight = mapped_weights.get(y_test[i], 1)\n    test_weighted_loss += -weight * np.log(max(prob, 1e-15))\n\ntest_weighted_loss /= sum(test_weights)\n\ntest_preds = np.argmax(ensemble_pred_test, axis=1)\ntest_accuracy = (y_test == test_preds).mean()\n\ntest_f1_macro = f1_score(y_test, test_preds, average='macro')\ntest_f1_weighted = f1_score(y_test, test_preds, average='weighted')\n\nprint(f\"Weighted Log Loss: {test_weighted_loss:.4f}\")\nprint(f\"Accuracy: {test_accuracy:.4f}\")\nprint(f\"Macro F1-score: {test_f1_macro:.4f}\")\nprint(f\"Weighted F1-score: {test_f1_weighted:.4f}\")\n\nprint(f\"\\nComparison to Kaggle 1st Place (Private LB: 0.51):\")\nprint(f\"Difference: {test_weighted_loss - 0.51:.4f} ({(test_weighted_loss - 0.51)/0.51 * 100:.1f}% higher)\")\n\nprint(\"\\nPer-class Test Set Performance:\")\nprint(classification_report(y_test, test_preds, target_names=[f\"Class {i}\" for i in range(n_classes)]))\n\nprint(\"\\nTest set evaluation completed\")","metadata":{"_uuid":"8668b84f-f01d-4356-8bb5-51946f950c4f","_cell_guid":"016f7e76-35e4-4d79-b65e-7f277f82435d","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 16: Save Results\nprint(\"\\nSaving results...\")\n\nsubmission_df = pd.DataFrame({\n    'object_id': features['object_id'],\n    'true_target': y,\n    'predicted_target': y_pred,\n    'true_class': features['target_original'],\n    'predicted_class': [reverse_mapping[p] for p in y_pred]\n})\n\nsubmission_df.to_csv('final_predictions.csv', index=False)\nprint(\"Predictions saved to 'final_predictions.csv'\")\n\nimportance_summary.to_csv('feature_importance_final.csv')\nprint(\"Feature importance saved to 'feature_importance_final.csv'\")\n\nresults_df.to_csv('model_comparison_final.csv', index=False)\nprint(\"Model comparison saved to 'model_comparison_final.csv'\")\n\nmodels[-1].save_model('best_lgb_model.txt')\nprint(\"Best LightGBM model saved to 'best_lgb_model.txt'\")","metadata":{"_uuid":"b8a4f141-d6e4-4a49-9493-247c075debb2","_cell_guid":"22d8e54e-1dbd-4554-aa5e-09269c02f642","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 17: Key Results Summary\nprint(\"\\nKey Results Summary\")\n\nprint(\"\\n1. Dataset:\")\nprint(f\"   Total objects: {len(features):,}\")\nprint(f\"   Number of features: {len(feature_cols)}\")\nprint(f\"   Number of classes: {len(class_mapping)}\")\n\nprint(\"\\n2. Model Performance:\")\nfor idx, row in results_df.iterrows():\n    print(f\"   {row['Algorithm']:15s}: Log Loss = {row['Log Loss']:.3f}, Accuracy = {row['Accuracy']:.3f}\")\n\nprint(\"\\n3. Top Features:\")\nfor i, (feat, imp) in enumerate(mean_importance.head(5).items(), 1):\n    print(f\"   {i}. {feat}: {imp:.2f}\")\n\nprint(\"\\n4. Computational Efficiency:\")\ntotal_time = sum([r['Time (s)'] for r in results_comparison if r['Time (s)'] > 0])\nprint(f\"   Total training time: {total_time:.1f} seconds\")","metadata":{"_uuid":"41251618-ff6b-4f13-8c50-5e8926cbf1ed","_cell_guid":"abc6dd4b-1b70-44a4-bca8-3088e0fcabc6","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Cell 18: Quick Prediction Demo\nprint(\"\\nQuick Prediction Demo\")\n\nsample_indices = np.random.choice(len(X), 5, replace=False)\n\nprint(\"\\nSample predictions:\")\nfor idx in sample_indices:\n    true_class = reverse_mapping[y[idx]]\n    pred_probs = predictions[idx]\n    pred_class = reverse_mapping[np.argmax(pred_probs)]\n    confidence = pred_probs[np.argmax(pred_probs)]\n    \n    print(f\"Object {features.iloc[idx]['object_id']}:\")\n    print(f\"  True class: {true_class}\")\n    print(f\"  Predicted: {pred_class} (confidence: {confidence:.3f})\")\n    print(f\"  Correct: {'Yes' if true_class == pred_class else 'No'}\")\n    print()\n\nprint(\"\\nAll processing completed successfully\")","metadata":{"_uuid":"7e134402-b151-4dc3-9dc2-1d8804ccd960","_cell_guid":"e41e0c28-79ee-423d-a62e-40eede74a002","trusted":true,"collapsed":false,"jupyter":{"outputs_hidden":false}},"outputs":[],"execution_count":null}]}