{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-20T09:54:21.417910Z","iopub.execute_input":"2023-06-20T09:54:21.419054Z","iopub.status.idle":"2023-06-20T09:54:21.458645Z","shell.execute_reply.started":"2023-06-20T09:54:21.418900Z","shell.execute_reply":"2023-06-20T09:54:21.457239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split, RandomizedSearchCV, GridSearchCV\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.svm import SVC\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nimport random\nimport gc\nfrom datatable import dt, f, ifelse, update, mean, by\nfrom sklearn.preprocessing import OneHotEncoder\nimport pickle\nimport catboost\n\nimport optuna\nfrom optuna.samplers import TPESampler\n\nrandom.seed(42)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:23:37.342453Z","iopub.execute_input":"2023-06-20T10:23:37.343019Z","iopub.status.idle":"2023-06-20T10:23:37.640162Z","shell.execute_reply.started":"2023-06-20T10:23:37.342971Z","shell.execute_reply":"2023-06-20T10:23:37.638905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = pd.read_parquet('/kaggle/input/amex-parquet/train_data.parquet')\n# df = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=50000)\n# y = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')\ntrain_dt = dt.fread('/kaggle/input/d/uom180259b/gene-expression/x_train.csv')\ntrain_dt.to_jay('train_data.jay')\n# test_features = pd.read_parquet('/kaggle/input/amex-parquet/test_data.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:05.077417Z","iopub.execute_input":"2023-06-20T12:42:05.077890Z","iopub.status.idle":"2023-06-20T12:42:05.274784Z","shell.execute_reply.started":"2023-06-20T12:42:05.077854Z","shell.execute_reply":"2023-06-20T12:42:05.272795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt = dt.fread('/kaggle/working/train_data.jay')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:05.285825Z","iopub.execute_input":"2023-06-20T12:42:05.286654Z","iopub.status.idle":"2023-06-20T12:42:05.301897Z","shell.execute_reply.started":"2023-06-20T12:42:05.286599Z","shell.execute_reply":"2023-06-20T12:42:05.299518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y = dt.fread('/kaggle/input/d/uom180259b/gene-expression/y_train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:05.461983Z","iopub.execute_input":"2023-06-20T12:42:05.462696Z","iopub.status.idle":"2023-06-20T12:42:05.475119Z","shell.execute_reply.started":"2023-06-20T12:42:05.462647Z","shell.execute_reply":"2023-06-20T12:42:05.472534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del y['Id']","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:05.630572Z","iopub.execute_input":"2023-06-20T12:42:05.631080Z","iopub.status.idle":"2023-06-20T12:42:05.638433Z","shell.execute_reply.started":"2023-06-20T12:42:05.631040Z","shell.execute_reply":"2023-06-20T12:42:05.636197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt[:,update(**{key: ifelse(f[key]==None,\n                              0, \n                              f[key]) \n    for key in train_dt.names})]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:06.330873Z","iopub.execute_input":"2023-06-20T12:42:06.331581Z","iopub.status.idle":"2023-06-20T12:42:06.340224Z","shell.execute_reply.started":"2023-06-20T12:42:06.331534Z","shell.execute_reply":"2023-06-20T12:42:06.338328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:06.955915Z","iopub.execute_input":"2023-06-20T12:42:06.956513Z","iopub.status.idle":"2023-06-20T12:42:06.966834Z","shell.execute_reply.started":"2023-06-20T12:42:06.956471Z","shell.execute_reply":"2023-06-20T12:42:06.965102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt_mean = train_dt[:, mean(f[:]), by('Id')]\ntrain_dt_std = train_dt[:, dt.sd(f[:]), by('Id')]\ntrain_dt_max = train_dt[:, dt.max(f[:]), by('Id')]\ntrain_dt_min = train_dt[:, dt.min(f[:]), by('Id')]\ntrain_dt_last = train_dt[:, dt.last(f[:]), by('Id')]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:07.906791Z","iopub.execute_input":"2023-06-20T12:42:07.907471Z","iopub.status.idle":"2023-06-20T12:42:08.848919Z","shell.execute_reply.started":"2023-06-20T12:42:07.907412Z","shell.execute_reply":"2023-06-20T12:42:08.847783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dt_mean['Id']\ndel train_dt_std['Id']\ndel train_dt_max['Id']\ndel train_dt_min['Id']\ndel train_dt_last['Id']","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:08.851753Z","iopub.execute_input":"2023-06-20T12:42:08.852520Z","iopub.status.idle":"2023-06-20T12:42:08.859778Z","shell.execute_reply.started":"2023-06-20T12:42:08.852474Z","shell.execute_reply":"2023-06-20T12:42:08.858389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt_mean.names = ['mean_'+key for key in train_dt_mean.names]\ntrain_dt_std.names = ['sd_'+key for key in train_dt_std.names]\ntrain_dt_max.names = ['max_'+key for key in train_dt_max.names]\ntrain_dt_min.names = ['min_'+key for key in train_dt_min.names]\ntrain_dt_last.names = ['last_'+key for key in train_dt_last.names]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:08.865836Z","iopub.execute_input":"2023-06-20T12:42:08.867696Z","iopub.status.idle":"2023-06-20T12:42:08.883444Z","shell.execute_reply.started":"2023-06-20T12:42:08.867632Z","shell.execute_reply":"2023-06-20T12:42:08.881631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt = dt.cbind(train_dt_mean, train_dt_std, train_dt_max, train_dt_min, train_dt_last)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:09.008529Z","iopub.execute_input":"2023-06-20T12:42:09.009868Z","iopub.status.idle":"2023-06-20T12:42:09.017948Z","shell.execute_reply.started":"2023-06-20T12:42:09.009800Z","shell.execute_reply":"2023-06-20T12:42:09.016083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:09.511622Z","iopub.execute_input":"2023-06-20T12:42:09.512089Z","iopub.status.idle":"2023-06-20T12:42:09.523433Z","shell.execute_reply.started":"2023-06-20T12:42:09.512054Z","shell.execute_reply":"2023-06-20T12:42:09.521948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_dt_mean\ndel train_dt_std\ndel train_dt_max\ndel train_dt_min\ndel train_dt_last\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:09.881933Z","iopub.execute_input":"2023-06-20T12:42:09.882519Z","iopub.status.idle":"2023-06-20T12:42:10.490148Z","shell.execute_reply.started":"2023-06-20T12:42:09.882476Z","shell.execute_reply":"2023-06-20T12:42:10.488549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dt[:,update(**{key: ifelse(f[key]==None,\n                              0, \n                              f[key]) \n    for key in train_dt.names})]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:10.493302Z","iopub.execute_input":"2023-06-20T12:42:10.493900Z","iopub.status.idle":"2023-06-20T12:42:10.508837Z","shell.execute_reply.started":"2023-06-20T12:42:10.493856Z","shell.execute_reply":"2023-06-20T12:42:10.507293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train_dt.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:10.685058Z","iopub.execute_input":"2023-06-20T12:42:10.686075Z","iopub.status.idle":"2023-06-20T12:42:10.745910Z","shell.execute_reply.started":"2023-06-20T12:42:10.686003Z","shell.execute_reply":"2023-06-20T12:42:10.744569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_orig = pd.read_csv(\"/kaggle/input/d/uom180259b/gene-expression/x_train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:11.614751Z","iopub.execute_input":"2023-06-20T12:42:11.615206Z","iopub.status.idle":"2023-06-20T12:42:12.392120Z","shell.execute_reply.started":"2023-06-20T12:42:11.615171Z","shell.execute_reply":"2023-06-20T12:42:12.390427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_np = train_df_orig.to_numpy()[:, 1:].reshape((15485, 500))\ntrain_df_new = pd.concat([train_df, pd.DataFrame(train_df_np)], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:12.394898Z","iopub.execute_input":"2023-06-20T12:42:12.395439Z","iopub.status.idle":"2023-06-20T12:42:12.864204Z","shell.execute_reply.started":"2023-06-20T12:42:12.395395Z","shell.execute_reply":"2023-06-20T12:42:12.862339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(train_df_new, \n                                                    y.to_numpy().ravel(),\n                                                    test_size=0.20,\n                                                    random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:42:12.936795Z","iopub.execute_input":"2023-06-20T12:42:12.937871Z","iopub.status.idle":"2023-06-20T12:42:13.120310Z","shell.execute_reply.started":"2023-06-20T12:42:12.937818Z","shell.execute_reply":"2023-06-20T12:42:13.118472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb = XGBClassifier()\nxgb.fit(X_train,y_train)\ny_pred = xgb.predict_proba(X_test)\nroc_auc_score(y_test, y_pred[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:17:41.336987Z","iopub.execute_input":"2023-06-20T10:17:41.337522Z","iopub.status.idle":"2023-06-20T10:17:55.158842Z","shell.execute_reply.started":"2023-06-20T10:17:41.337486Z","shell.execute_reply":"2023-06-20T10:17:55.157676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"roc_auc_score(y_test, y_pred[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:18:25.710350Z","iopub.execute_input":"2023-06-20T10:18:25.710816Z","iopub.status.idle":"2023-06-20T10:18:25.723098Z","shell.execute_reply.started":"2023-06-20T10:18:25.710779Z","shell.execute_reply":"2023-06-20T10:18:25.721852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb.save_model('model-3.json')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'subsample': 0.6, 'n_estimators': 800, 'min_child_weight': 1, 'max_depth': 11, 'learning_rate': 0.15, 'colsample_bytree': 0.9}\nlgbm = LGBMClassifier(boosting_type='dart', **params)\nlgbm.fit(X_train,y_train)\ny_pred = lgbm.predict_proba(X_test)\nroc_auc_score(y_test, y_pred[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:19:55.627585Z","iopub.execute_input":"2023-06-20T10:19:55.628126Z","iopub.status.idle":"2023-06-20T10:20:24.943524Z","shell.execute_reply.started":"2023-06-20T10:19:55.628088Z","shell.execute_reply":"2023-06-20T10:20:24.942474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# CatBoost","metadata":{}},{"cell_type":"code","source":"model = catboost.CatBoostClassifier(verbose=False)\nmodel.fit(X_train, y_train)\ny_pred = model.predict_proba(X_test)\nroc_auc_score(y_test, y_pred[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:22:01.453261Z","iopub.execute_input":"2023-06-20T10:22:01.453757Z","iopub.status.idle":"2023-06-20T10:22:35.591459Z","shell.execute_reply.started":"2023-06-20T10:22:01.453721Z","shell.execute_reply":"2023-06-20T10:22:35.590159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def objective(trial):\n    model = catboost.CatBoostClassifier(\n        iterations=trial.suggest_int(\"iterations\", 100, 1000),\n        learning_rate=trial.suggest_float(\"learning_rate\", 1e-3, 1e-1, log=True),\n        depth=trial.suggest_int(\"depth\", 4, 10),\n        l2_leaf_reg=trial.suggest_float(\"l2_leaf_reg\", 1e-8, 100.0, log=True),\n        bootstrap_type=trial.suggest_categorical(\"bootstrap_type\", [\"Bayesian\"]),\n        random_strength=trial.suggest_float(\"random_strength\", 1e-8, 10.0, log=True),\n        bagging_temperature=trial.suggest_float(\"bagging_temperature\", 0.0, 10.0),\n        od_type=trial.suggest_categorical(\"od_type\", [\"IncToDec\", \"Iter\"]),\n        od_wait=trial.suggest_int(\"od_wait\", 10, 50),\n        verbose=False\n    )\n    model.fit(X_train, y_train)\n    y_pred = model.predict_proba(X_test)\n    return roc_auc_score(y_test, y_pred[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:25:54.161947Z","iopub.execute_input":"2023-06-20T10:25:54.162438Z","iopub.status.idle":"2023-06-20T10:25:54.173985Z","shell.execute_reply.started":"2023-06-20T10:25:54.162400Z","shell.execute_reply":"2023-06-20T10:25:54.172419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.logging.set_verbosity(optuna.logging.WARNING)\n\nsampler = TPESampler(seed=1)\nstudy = optuna.create_study(study_name=\"catboost\", direction=\"maximize\", sampler=sampler)\nstudy.optimize(objective, n_trials=100)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T10:26:17.975011Z","iopub.execute_input":"2023-06-20T10:26:17.975540Z","iopub.status.idle":"2023-06-20T11:19:16.628457Z","shell.execute_reply.started":"2023-06-20T10:26:17.975499Z","shell.execute_reply":"2023-06-20T11:19:16.627091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study.best_trial.params","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:19:16.790976Z","iopub.execute_input":"2023-06-20T11:19:16.792237Z","iopub.status.idle":"2023-06-20T11:19:16.800762Z","shell.execute_reply.started":"2023-06-20T11:19:16.792189Z","shell.execute_reply":"2023-06-20T11:19:16.799465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'iterations': 755,\n 'learning_rate': 0.05220395990340396,\n 'depth': 6,\n 'l2_leaf_reg': 3.9402056286448177,\n 'bootstrap_type': 'Bayesian',\n 'random_strength': 1.0680478181115725e-08,\n 'bagging_temperature': 9.643533501914009,\n 'od_type': 'IncToDec',\n 'od_wait': 45}\nmodel = catboost.CatBoostClassifier(**params, verbose=False)\nmodel.fit(X_train, y_train)\ny_pred = model.predict_proba(X_test)\nroc_auc_score(y_test, y_pred[:, 1])","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:39:25.734664Z","iopub.execute_input":"2023-06-20T11:39:25.735186Z","iopub.status.idle":"2023-06-20T11:39:56.977144Z","shell.execute_reply.started":"2023-06-20T11:39:25.735148Z","shell.execute_reply":"2023-06-20T11:39:56.975715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lgbm.booster_.save_model('lgb-model-2.txt')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parameters = {'learning_rate': [0.8],\n              'max_depth': [13],\n              'min_child_weight': [3],\n              'subsample': [0.7],\n              'colsample_bytree': [0.8],\n              'n_estimators': [1000],\n             'tree_method': ['gpu_hist']}\n# {'tree_method': 'gpu_hist', 'subsample': 0.7, 'n_estimators': 1000, 'min_child_weight': 3, 'max_depth': 13, 'learning_rate': 0.1, 'colsample_bytree': 0.8}","metadata":{"execution":{"iopub.status.busy":"2023-01-21T16:00:40.528954Z","iopub.execute_input":"2023-01-21T16:00:40.529331Z","iopub.status.idle":"2023-01-21T16:00:40.534596Z","shell.execute_reply.started":"2023-01-21T16:00:40.529296Z","shell.execute_reply":"2023-01-21T16:00:40.533619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XGBoost","metadata":{}},{"cell_type":"code","source":"def objective(trial):\n    \"\"\"Define the objective function\"\"\"\n\n    params = {\n        'max_depth': trial.suggest_int('max_depth', 1, 9),\n        'learning_rate': trial.suggest_float('learning_rate', 0.01, 1.0),\n        'n_estimators': trial.suggest_int('n_estimators', 50, 500),\n        'min_child_weight': trial.suggest_int('min_child_weight', 1, 10),\n        'gamma': trial.suggest_float('gamma', 1e-8, 1.0),\n        'subsample': trial.suggest_float('subsample', 0.01, 1.0),\n        'colsample_bytree': trial.suggest_float('colsample_bytree', 0.01, 1.0),\n        'reg_alpha': trial.suggest_float('reg_alpha', 1e-8, 1.0),\n        'reg_lambda': trial.suggest_float('reg_lambda', 1e-8, 1.0),\n        'eval_metric': 'mlogloss',\n        'use_label_encoder': False\n    }\n\n    # Fit the model\n    optuna_model = XGBClassifier(**params)\n    optuna_model.fit(X_train, y_train)\n\n    # Make predictions\n    y_pred = optuna_model.predict_proba(X_test)\n\n    # Evaluate predictions\n    accuracy = roc_auc_score(y_test, y_pred[:, 1])\n    return accuracy","metadata":{"execution":{"iopub.status.busy":"2023-06-20T11:40:43.593322Z","iopub.execute_input":"2023-06-20T11:40:43.593794Z","iopub.status.idle":"2023-06-20T11:40:43.605947Z","shell.execute_reply.started":"2023-06-20T11:40:43.593759Z","shell.execute_reply":"2023-06-20T11:40:43.604582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.logging.set_verbosity(optuna.logging.WARNING)\n\nsampler = TPESampler(seed=1)\nstudy = optuna.create_study(study_name=\"catboost\", direction=\"maximize\", sampler=sampler)\nstudy.optimize(objective, n_trials=100)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(study.best_trial)\nprint(study.best_trial.params)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:23:00.002788Z","iopub.execute_input":"2023-06-20T12:23:00.003347Z","iopub.status.idle":"2023-06-20T12:23:00.011556Z","shell.execute_reply.started":"2023-06-20T12:23:00.003306Z","shell.execute_reply":"2023-06-20T12:23:00.010157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {'max_depth': 8, 'learning_rate': 0.01488312820698747, 'n_estimators': 404, 'min_child_weight': 10, 'gamma': 0.0008354617686371871, 'subsample': 0.9763697242710973, 'colsample_bytree': 0.0600342244421979, 'reg_alpha': 1.0054814184244183e-05, 'reg_lambda': 0.00029265479545570575}\nmodels = []\nfor i in range(10):\n    np.random.seed(i)\n    model = XGBClassifier(**params, seed=i, random_state=i)\n    model.fit(train_df_new, y.to_numpy().ravel())\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:54:25.091744Z","iopub.execute_input":"2023-06-20T12:54:25.092306Z","iopub.status.idle":"2023-06-20T12:57:27.624261Z","shell.execute_reply.started":"2023-06-20T12:54:25.092268Z","shell.execute_reply":"2023-06-20T12:57:27.623241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"test_dt = dt.fread('/kaggle/input/d/uom180259b/gene-expression/x_test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:21.935252Z","iopub.execute_input":"2023-06-20T12:44:21.935776Z","iopub.status.idle":"2023-06-20T12:44:21.965340Z","shell.execute_reply.started":"2023-06-20T12:44:21.935734Z","shell.execute_reply":"2023-06-20T12:44:21.964018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:22.159083Z","iopub.execute_input":"2023-06-20T12:44:22.159540Z","iopub.status.idle":"2023-06-20T12:44:22.168732Z","shell.execute_reply.started":"2023-06-20T12:44:22.159503Z","shell.execute_reply":"2023-06-20T12:44:22.166895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt[:,update(**{key: ifelse(f[key]==None,\n                              0, \n                              f[key]) \n    for key in test_dt.names})]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:22.376849Z","iopub.execute_input":"2023-06-20T12:44:22.377285Z","iopub.status.idle":"2023-06-20T12:44:22.385241Z","shell.execute_reply.started":"2023-06-20T12:44:22.377252Z","shell.execute_reply":"2023-06-20T12:44:22.383528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_features.fillna(0, inplace=True)\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:22.590110Z","iopub.execute_input":"2023-06-20T12:44:22.590619Z","iopub.status.idle":"2023-06-20T12:44:23.128062Z","shell.execute_reply.started":"2023-06-20T12:44:22.590583Z","shell.execute_reply":"2023-06-20T12:44:23.127053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt_mean = test_dt[:, mean(f[:]), by('Id')]\ntest_dt_std = test_dt[:, dt.sd(f[:]), by('Id')]\ntest_dt_max = test_dt[:, dt.max(f[:]), by('Id')]\ntest_dt_min = test_dt[:, dt.min(f[:]), by('Id')]\ntest_dt_last = test_dt[:, dt.last(f[:]), by('Id')]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:23.130539Z","iopub.execute_input":"2023-06-20T12:44:23.131512Z","iopub.status.idle":"2023-06-20T12:44:23.362689Z","shell.execute_reply.started":"2023-06-20T12:44:23.131472Z","shell.execute_reply":"2023-06-20T12:44:23.361649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dt_mean['Id']\ndel test_dt_std['Id']\ndel test_dt_max['Id']\ndel test_dt_min['Id']\ndel test_dt_last['Id']\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:23.364286Z","iopub.execute_input":"2023-06-20T12:44:23.368901Z","iopub.status.idle":"2023-06-20T12:44:23.783011Z","shell.execute_reply.started":"2023-06-20T12:44:23.368853Z","shell.execute_reply":"2023-06-20T12:44:23.781831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt_mean.names = ['mean_'+key for key in test_dt_mean.names]\ntest_dt_std.names = ['sd_'+key for key in test_dt_std.names]\ntest_dt_max.names = ['max_'+key for key in test_dt_max.names]\ntest_dt_min.names = ['min_'+key for key in test_dt_min.names]\ntest_dt_last.names = ['last_'+key for key in test_dt_last.names]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:23.785297Z","iopub.execute_input":"2023-06-20T12:44:23.785699Z","iopub.status.idle":"2023-06-20T12:44:23.794652Z","shell.execute_reply.started":"2023-06-20T12:44:23.785665Z","shell.execute_reply":"2023-06-20T12:44:23.793442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt = dt.cbind(test_dt_mean, test_dt_std, test_dt_max, test_dt_min, test_dt_last)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:23.796080Z","iopub.execute_input":"2023-06-20T12:44:23.796439Z","iopub.status.idle":"2023-06-20T12:44:23.808998Z","shell.execute_reply.started":"2023-06-20T12:44:23.796407Z","shell.execute_reply":"2023-06-20T12:44:23.807772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:23.938932Z","iopub.execute_input":"2023-06-20T12:44:23.939392Z","iopub.status.idle":"2023-06-20T12:44:23.948325Z","shell.execute_reply.started":"2023-06-20T12:44:23.939331Z","shell.execute_reply":"2023-06-20T12:44:23.946967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_dt_mean\ndel test_dt_std\ndel test_dt_max\ndel test_dt_min\ndel test_dt_last\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:24.348197Z","iopub.execute_input":"2023-06-20T12:44:24.348665Z","iopub.status.idle":"2023-06-20T12:44:24.868470Z","shell.execute_reply.started":"2023-06-20T12:44:24.348630Z","shell.execute_reply":"2023-06-20T12:44:24.866896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dt[:,update(**{key: ifelse(f[key]==None,\n                              0, \n                              f[key]) \n    for key in test_dt.names})]","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:24.871095Z","iopub.execute_input":"2023-06-20T12:44:24.871591Z","iopub.status.idle":"2023-06-20T12:44:24.881849Z","shell.execute_reply.started":"2023-06-20T12:44:24.871551Z","shell.execute_reply":"2023-06-20T12:44:24.880704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = test_dt.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:25.511161Z","iopub.execute_input":"2023-06-20T12:44:25.511696Z","iopub.status.idle":"2023-06-20T12:44:25.531229Z","shell.execute_reply.started":"2023-06-20T12:44:25.511655Z","shell.execute_reply":"2023-06-20T12:44:25.529942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(test_df.columns)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:26.169027Z","iopub.execute_input":"2023-06-20T12:44:26.169516Z","iopub.status.idle":"2023-06-20T12:44:26.176682Z","shell.execute_reply.started":"2023-06-20T12:44:26.169480Z","shell.execute_reply":"2023-06-20T12:44:26.175737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:26.469924Z","iopub.execute_input":"2023-06-20T12:44:26.470428Z","iopub.status.idle":"2023-06-20T12:44:26.891720Z","shell.execute_reply.started":"2023-06-20T12:44:26.470361Z","shell.execute_reply":"2023-06-20T12:44:26.890182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_orig = pd.read_csv(\"/kaggle/input/d/uom180259b/gene-expression/x_test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:26.894534Z","iopub.execute_input":"2023-06-20T12:44:26.894919Z","iopub.status.idle":"2023-06-20T12:44:27.068547Z","shell.execute_reply.started":"2023-06-20T12:44:26.894888Z","shell.execute_reply":"2023-06-20T12:44:27.067095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df_orig = test_df_orig.fillna(0)\ntest_df_np = test_df_orig.to_numpy()[:, 1:].reshape((3871, 500))\ntest_df_new = pd.concat([test_df, pd.DataFrame(test_df_np)], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:44:27.558556Z","iopub.execute_input":"2023-06-20T12:44:27.559053Z","iopub.status.idle":"2023-06-20T12:44:27.603967Z","shell.execute_reply.started":"2023-06-20T12:44:27.559015Z","shell.execute_reply":"2023-06-20T12:44:27.602447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = models[0].predict_proba(test_df_new)[:, 1]\nfor model in models[1:]:\n    y_pred += model.predict_proba(test_df_new)[:, 1]\ny_pred /= len(models)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T12:57:27.626205Z","iopub.execute_input":"2023-06-20T12:57:27.627215Z","iopub.status.idle":"2023-06-20T12:57:28.463426Z","shell.execute_reply.started":"2023-06-20T12:57:27.627176Z","shell.execute_reply":"2023-06-20T12:57:28.462448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"GeneId\": test_df_orig[\"Id\"].unique(), \"Prediction\": y_pred})","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:00:55.101218Z","iopub.execute_input":"2023-06-20T13:00:55.101747Z","iopub.status.idle":"2023-06-20T13:00:55.113122Z","shell.execute_reply.started":"2023-06-20T13:00:55.101710Z","shell.execute_reply":"2023-06-20T13:00:55.112019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:00:55.377018Z","iopub.execute_input":"2023-06-20T13:00:55.378527Z","iopub.status.idle":"2023-06-20T13:00:55.390535Z","shell.execute_reply.started":"2023-06-20T13:00:55.378467Z","shell.execute_reply":"2023-06-20T13:00:55.389184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission_3.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-20T13:01:04.515878Z","iopub.execute_input":"2023-06-20T13:01:04.516581Z","iopub.status.idle":"2023-06-20T13:01:04.536707Z","shell.execute_reply.started":"2023-06-20T13:01:04.516544Z","shell.execute_reply":"2023-06-20T13:01:04.535312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}