{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport json\n\nimport catboost as cb\nimport os\nimport gc\nfrom tqdm import tqdm\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import defaultdict\nfrom IPython.display import FileLink        \nimport pickle as pkl\nfrom sklearn.model_selection import train_test_split\n\nimport matplotlib.gridspec as gridspec\nimport lightgbm as lgmb\nfrom scipy.stats import spearmanr\nimport scipy","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-28T22:39:52.827403Z","iopub.execute_input":"2023-02-28T22:39:52.828291Z","iopub.status.idle":"2023-02-28T22:39:52.836177Z","shell.execute_reply.started":"2023-02-28T22:39:52.828226Z","shell.execute_reply":"2023-02-28T22:39:52.834728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RES_BASE_DIR = \"/kaggle/input/catboost-models-v7/models\"","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.883310Z","iopub.execute_input":"2023-02-28T22:33:57.883672Z","iopub.status.idle":"2023-02-28T22:33:57.890036Z","shell.execute_reply.started":"2023-02-28T22:33:57.883637Z","shell.execute_reply":"2023-02-28T22:33:57.888045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def correlation_score(y_true, y_pred):\n    \"\"\"Scores the predictions according to the competition rules. \n    \n    It is assumed that the predictions are not constant.\n    \n    Returns the average of each sample's Pearson correlation coefficient\"\"\"\n    \n    y2 = y_pred.copy()\n    y2 -= y2.mean(axis=0);    y2 /= y2.std(axis=0) \n    y1 = y_true.copy(); \n    y1 -= y1.mean(axis=0);    y1 /= y1.std(axis=0) \n        \n    c = (y1*y2).mean().mean()# Correlation for rescaled matrices is just matrix product and average \n        \n    return c","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.891910Z","iopub.execute_input":"2023-02-28T22:33:57.892441Z","iopub.status.idle":"2023-02-28T22:33:57.907368Z","shell.execute_reply.started":"2023-02-28T22:33:57.892388Z","shell.execute_reply":"2023-02-28T22:33:57.905779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def normalization(values):\n    return (values - np.mean(values)) / np.std(values)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.911147Z","iopub.execute_input":"2023-02-28T22:33:57.911970Z","iopub.status.idle":"2023-02-28T22:33:57.920537Z","shell.execute_reply.started":"2023-02-28T22:33:57.911913Z","shell.execute_reply":"2023-02-28T22:33:57.919293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def intersaction_coeff(list1, list2):\n    return len(set(list1).intersection(set(list2)))","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.922112Z","iopub.execute_input":"2023-02-28T22:33:57.922607Z","iopub.status.idle":"2023-02-28T22:33:57.933943Z","shell.execute_reply.started":"2023-02-28T22:33:57.922555Z","shell.execute_reply":"2023-02-28T22:33:57.932470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cos_sim(a, b):\n    cos_sim = np.dot(a, b)/(np.linalg.norm(a)*np.linalg.norm(b))\n    return cos_sim","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.935211Z","iopub.execute_input":"2023-02-28T22:33:57.935615Z","iopub.status.idle":"2023-02-28T22:33:57.948951Z","shell.execute_reply.started":"2023-02-28T22:33:57.935561Z","shell.execute_reply":"2023-02-28T22:33:57.947063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def open_json(path):\n    with open(path, \"r\") as json_file:\n        data = json.load(json_file)\n        return data","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.951437Z","iopub.execute_input":"2023-02-28T22:33:57.951965Z","iopub.status.idle":"2023-02-28T22:33:57.962258Z","shell.execute_reply.started":"2023-02-28T22:33:57.951908Z","shell.execute_reply":"2023-02-28T22:33:57.960622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = open_json(f\"{RES_BASE_DIR}/msci_config_experiment.json\")\nlgbm_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_all_lgbm.feather\")\nlgbm_dupl_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_dupl_all_lgbm.feather\")\nlgbm_perm_df = pd.read_feather(f\"{RES_BASE_DIR}/fi_perm_all_lgbm.feather\")\n\nduplicate_columns = config[\"duplicate_columns\"]\npermutate_columns = config[\"permutate_columns\"]\ncolumns = config[\"columns\"]\nbatches_config = config[\"batches\"]\n\nfi_lgbm_repated_df = pd.read_pickle(\"/kaggle/input/fi-repeated/fi_perm_batch_22.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:57.964726Z","iopub.execute_input":"2023-02-28T22:33:57.965650Z","iopub.status.idle":"2023-02-28T22:33:59.518096Z","shell.execute_reply.started":"2023-02-28T22:33:57.965551Z","shell.execute_reply":"2023-02-28T22:33:59.516669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sr = fi_lgbm_repated_df.iloc[0]","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:59.519798Z","iopub.execute_input":"2023-02-28T22:33:59.520647Z","iopub.status.idle":"2023-02-28T22:33:59.528200Z","shell.execute_reply.started":"2023-02-28T22:33:59.520605Z","shell.execute_reply":"2023-02-28T22:33:59.526963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sr","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:59.533126Z","iopub.execute_input":"2023-02-28T22:33:59.533936Z","iopub.status.idle":"2023-02-28T22:33:59.546184Z","shell.execute_reply.started":"2023-02-28T22:33:59.533885Z","shell.execute_reply":"2023-02-28T22:33:59.545206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_fi(\n    fi,        \n    prefix=\"PERM\",\n    N = 20,\n    title = None\n):\n    if prefix == \"DUPL\":\n        columns = duplicate_columns\n    elif prefix == \"PERM\":\n        columns = permutate_columns\n    else:\n        raise \"Not supported prefix\"\n    \n    sorted_index = np.argsort(fi)[::-1]\n    sorted_fi = fi[sorted_index]\n    sorted_columns = np.array(columns)[sorted_index]    \n    \n    vanil_indexes = []\n    no_vanil_indexes = []\n    for idx, column_name in enumerate(sorted_columns[:N]):\n        if column_name.startswith(prefix):\n            no_vanil_indexes.append(idx)\n        else:\n            vanil_indexes.append(idx)   \n            \n    _ = plt.scatter(\n        vanil_indexes, \n        sorted_fi[vanil_indexes],\n        label=\"vanil\"\n    )    \n\n    _ = plt.scatter(\n        no_vanil_indexes, \n        sorted_fi[no_vanil_indexes],\n        s=8,\n        #alpha=0.5,\n        label=prefix,\n        color=\"orange\"\n    )     \n    _ = plt.xlabel(\"Feature RANK\")\n    _ = plt.ylabel(\"Feature score\")\n    if title is not None:\n        _ = plt.title(title)\n    \n    plt.legend()\n    return sorted_columns[:N], vanil_indexes, no_vanil_indexes","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:59.548160Z","iopub.execute_input":"2023-02-28T22:33:59.548831Z","iopub.status.idle":"2023-02-28T22:33:59.564986Z","shell.execute_reply.started":"2023-02-28T22:33:59.548786Z","shell.execute_reply":"2023-02-28T22:33:59.563919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sorted_columns, vanil_indexes, no_vanil_indexes = plot_fi(\n    sr.fi,\n    N=100\n)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:59.566459Z","iopub.execute_input":"2023-02-28T22:33:59.567095Z","iopub.status.idle":"2023-02-28T22:33:59.926332Z","shell.execute_reply.started":"2023-02-28T22:33:59.567046Z","shell.execute_reply":"2023-02-28T22:33:59.925288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_cite_input_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\")\ntrain_cite_target_df = pd.read_hdf(\"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\")\nmetadata_df = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")\n\nmetadata_df.set_index(\"cell_id\", inplace=True)\ntrain_cite_target_df = train_cite_target_df.join(metadata_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:33:59.927757Z","iopub.execute_input":"2023-02-28T22:33:59.928382Z","iopub.status.idle":"2023-02-28T22:35:08.867757Z","shell.execute_reply.started":"2023-02-28T22:33:59.928339Z","shell.execute_reply":"2023-02-28T22:35:08.866357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_df = defaultdict(list)\ncd_it = sr.CD\nday_it = sr.day\ndonor_it = sr.donor\n\ntrain_test_index = train_cite_target_df[\n    (train_cite_target_df.day == day_it) & \n    (train_cite_target_df.donor == donor_it)\n].index.tolist()        \n\ntrain_test_input_df = train_cite_input_df.loc[train_test_index]\nfor column in columns:\n    np.random.seed(42)\n    values = train_test_input_df[column].values.copy()\n    np.random.shuffle(values)\n    train_test_input_df[f\"PERM_{column}\"] = values\n\ntrain_test_input_df = train_test_input_df[permutate_columns]\n\ntrain_index, test_index = train_test_split(\n    train_test_index, \n    test_size=0.33, \n    random_state=42\n)\n\ntrain_X = train_test_input_df.loc[train_index]\ntrain_Y = train_cite_target_df.loc[train_index][cd_it]\n\ntest_X = train_test_input_df.loc[test_index]\ntest_Y = train_cite_target_df.loc[test_index][cd_it].values\n\n# del train_test_input_df\n# del y_true\n# del y_pred\n# del model\n# del train_index\n# del test_index\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:35:08.869331Z","iopub.execute_input":"2023-02-28T22:35:08.869751Z","iopub.status.idle":"2023-02-28T22:37:30.623834Z","shell.execute_reply.started":"2023-02-28T22:35:08.869705Z","shell.execute_reply":"2023-02-28T22:37:30.622622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrs = []\nbad_vanil_corrs_columns = []\nfor column in sorted_columns:\n    values_X = test_X[column].values\n    test_corr = correlation_score(\n        values_X,\n        test_Y\n    )\n    corrs.append(test_corr)\n    if not column.startswith(\"PERM\"):\n        perm_column = \"PERM_\" + column\n        perm_values_X = test_X[perm_column].values\n        perm_test_corr = correlation_score(\n            perm_values_X,\n            test_Y\n        )        \n        bad_vanil_corrs_columns.append({\n            \"column\": column,\n            \"perm_column\": perm_column,\n            \"corr\": test_corr,\n            \"perm_corr\": perm_test_corr,\n            \"spearmanr\": spearmanr(values_X[:], test_Y[:]).correlation,\n            \"perm_spearmanr\": spearmanr(perm_values_X[:], test_Y[:]).correlation\n        })\n    \n    #print(column, test_corr)\n_ = plt.title(\"Correlation\")   \n_ = plt.plot(sorted_columns, corrs, label=\"vanil\")\n_ = plt.scatter(\n    no_vanil_indexes, \n    np.array(corrs)[no_vanil_indexes], \n    color=\"orange\",\n    label=\"PERM\"\n)\n_ = plt.xticks(rotation=90)\n_ = plt.legend()\n_ = plt.xlabel(\"RNA\")\n_ = plt.ylabel(\"Corr(test_x, test_y)\")","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:37:30.625362Z","iopub.execute_input":"2023-02-28T22:37:30.625743Z","iopub.status.idle":"2023-02-28T22:37:33.349047Z","shell.execute_reply.started":"2023-02-28T22:37:30.625708Z","shell.execute_reply":"2023-02-28T22:37:33.347460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corrs = []\n# bad_vanil_corrs_columns = []\n# for column in sorted_columns:\n#     values_X = test_X[column].values\n#     #print(values_X.shape, test_Y.shape)\n#     test_corr = spearmanr(\n#         values_X[:],\n#         test_Y[:]\n#     ).correlation\n    \n#     if not column.startswith(\"PERM\") and np.abs(test_corr) < 0.1:\n#         bad_vanil_corrs_columns.append({\n#             \"column\": column,\n#             \"corr\": test_corr\n#         })\n        \n#     corrs.append(test_corr)\n#     #print(column, test_corr)\n# _ = plt.title(\"Spearmanr correlation\")   \n# _ = plt.plot(sorted_columns, corrs, label=\"vanil\")\n# _ = plt.scatter(\n#     no_vanil_indexes, \n#     np.array(corrs)[no_vanil_indexes], \n#     color=\"orange\",\n#     label=\"PERM\"\n# )\n# _ = plt.xticks(rotation=90)\n# _ = plt.legend()\n# _ = plt.xlabel(\"RNA\")\n# _ = plt.ylabel(\"Corr(test_x, test_y)\")","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:37:33.350686Z","iopub.execute_input":"2023-02-28T22:37:33.351919Z","iopub.status.idle":"2023-02-28T22:37:33.357941Z","shell.execute_reply.started":"2023-02-28T22:37:33.351862Z","shell.execute_reply":"2023-02-28T22:37:33.356519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sr","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:37:33.359537Z","iopub.execute_input":"2023-02-28T22:37:33.359996Z","iopub.status.idle":"2023-02-28T22:37:33.379016Z","shell.execute_reply.started":"2023-02-28T22:37:33.359947Z","shell.execute_reply":"2023-02-28T22:37:33.377694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corrs = [it[\"corr\"] for it in bad_vanil_corrs_columns]\nperm_corrs = [it[\"perm_corr\"] for it in bad_vanil_corrs_columns]\n\nspearmanr = [it[\"spearmanr\"] for it in bad_vanil_corrs_columns]\nperm_spearmanr = [it[\"perm_spearmanr\"] for it in bad_vanil_corrs_columns]","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:37:33.380825Z","iopub.execute_input":"2023-02-28T22:37:33.381464Z","iopub.status.idle":"2023-02-28T22:37:33.392766Z","shell.execute_reply.started":"2023-02-28T22:37:33.381425Z","shell.execute_reply":"2023-02-28T22:37:33.391470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, (ax1, ax2) = plt.subplots(1, 2, figsize=(10, 4))\n#fig.set_size(figsize=(8, 6), dpi=80)\n\n_ = plt.suptitle(f\"{sr.CD}_{sr.day}_{sr.donor}\")\n_ = ax1.set_ylabel(\"Corr\")\n_ = ax1.plot(corrs, label=\"vanil\")\n_ = ax1.plot(perm_corrs, label=\"perm\")\n_ = ax1.legend()\n\n_ = ax2.set_ylabel(\"Corr spearmanr\")\n_ = ax2.plot(spearmanr, label=\"vanil\")\n_ = ax2.plot(perm_spearmanr, label=\"perm\")\n_ = ax2.legend()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:37:33.393987Z","iopub.execute_input":"2023-02-28T22:37:33.394881Z","iopub.status.idle":"2023-02-28T22:37:33.787294Z","shell.execute_reply.started":"2023-02-28T22:37:33.394843Z","shell.execute_reply":"2023-02-28T22:37:33.785668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fi_df = defaultdict(list)\n\nfor it in tqdm(bad_vanil_corrs_columns):\n    for column in [it[\"column\"], it[\"perm_column\"]]:\n        train_x = train_X[column].values\n        train_y = train_Y\n        \n        test_x = test_X[column].values\n        test_y = test_Y\n        \n        model = lgmb.LGBMRegressor(\n            verbose=0,\n            random_state=42\n        )\n\n        model.fit(\n            np.expand_dims(train_x, 1),\n            train_y\n        )\n        \n        model_dir_path = f\"/kaggle/working/lgmb_{column}_{sr.day}_{sr.donor}\"    \n        os.makedirs(model_dir_path, exist_ok=True)\n\n        model_path = model_dir_path + \"/model.txt\"\n        model.booster_.save_model(model_path)\n\n        y_pred = model.predict(\n            np.expand_dims(test_x, 1)\n        )    \n        \n        corr_score = correlation_score(test_y, y_pred)\n        print(f\"{column}\", corr_score)\n\n        fi_df[\"CD\"].append(sr.CD)\n        fi_df[\"day\"].append(sr.day)\n        fi_df[\"donor\"].append(sr.donor)\n        fi_df[\"column\"].append(column)\n        fi_df[\"corr\"].append(corr_score)\n        \n#         break\n#     break\n    \nfi_df = pd.DataFrame(fi_df)\nfi_df.to_pickle(f\"/kaggle/working/fi_perm_single.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:37:33.789075Z","iopub.execute_input":"2023-02-28T22:37:33.789610Z","iopub.status.idle":"2023-02-28T22:38:47.575417Z","shell.execute_reply.started":"2023-02-28T22:37:33.789556Z","shell.execute_reply":"2023-02-28T22:38:47.574015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(fi_df)):\n    ef_i = i // 2\n    lgbm_column = fi_df.iloc[i].column\n    lgbm_corr = fi_df.iloc[i][\"corr\"]\n    \n    \n    assert (bad_vanil_corrs_columns[ef_i][\"column\"] == lgbm_column) or (bad_vanil_corrs_columns[ef_i][\"perm_column\"] == lgbm_column)\n    \n    if lgbm_column.startswith(\"PERM\"):\n        bad_vanil_corrs_columns[ef_i][\"perm_lgbm_corr\"] = lgbm_corr\n    else:\n        bad_vanil_corrs_columns[ef_i][\"lgbm_corr\"] = lgbm_corr\n    \n    bad_vanil_corrs_columns[ef_i][\"rank\"] = ef_i\n        \n    #print(ef_i)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:38:47.577587Z","iopub.execute_input":"2023-02-28T22:38:47.578059Z","iopub.status.idle":"2023-02-28T22:38:47.630700Z","shell.execute_reply.started":"2023-02-28T22:38:47.578009Z","shell.execute_reply":"2023-02-28T22:38:47.629533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lgbm_corrs = [it[\"lgbm_corr\"] for it in bad_vanil_corrs_columns]\nperm_lgbm_corrs = [it[\"perm_lgbm_corr\"] for it in bad_vanil_corrs_columns]","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:38:47.631972Z","iopub.execute_input":"2023-02-28T22:38:47.632889Z","iopub.status.idle":"2023-02-28T22:38:47.637694Z","shell.execute_reply.started":"2023-02-28T22:38:47.632850Z","shell.execute_reply":"2023-02-28T22:38:47.636710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = plt.title(\"Model on single feature\")\n_ = plt.plot(lgbm_corrs, label=\"vanil\")\n_ = plt.plot(perm_lgbm_corrs, label=\"perm\")\n_ = plt.legend()","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:38:47.639002Z","iopub.execute_input":"2023-02-28T22:38:47.639605Z","iopub.status.idle":"2023-02-28T22:38:47.896303Z","shell.execute_reply.started":"2023-02-28T22:38:47.639569Z","shell.execute_reply":"2023-02-28T22:38:47.895213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(bad_vanil_corrs_columns)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:38:47.897802Z","iopub.execute_input":"2023-02-28T22:38:47.898475Z","iopub.status.idle":"2023-02-28T22:38:47.905028Z","shell.execute_reply.started":"2023-02-28T22:38:47.898420Z","shell.execute_reply":"2023-02-28T22:38:47.903637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_corr_df = df[np.abs(df[\"corr\"]) < 0.1]\nlow_corr_df.sort_values(\"lgbm_corr\", inplace=True, ascending=False)\n\nlow_corr_df[[\n    \"rank\",\n    \"column\",    \n    \"corr\",\n    \"spearmanr\",\n    \"lgbm_corr\"\n]]","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:42:33.581698Z","iopub.execute_input":"2023-02-28T22:42:33.582212Z","iopub.status.idle":"2023-02-28T22:42:33.588166Z","shell.execute_reply.started":"2023-02-28T22:42:33.582170Z","shell.execute_reply":"2023-02-28T22:42:33.586659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_pickle(\"/kaggle/working/fi_perm_single.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:38:47.951729Z","iopub.execute_input":"2023-02-28T22:38:47.952228Z","iopub.status.idle":"2023-02-28T22:38:47.959874Z","shell.execute_reply.started":"2023-02-28T22:38:47.952177Z","shell.execute_reply":"2023-02-28T22:38:47.958435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for it in bad_vanil_corrs_columns:\n    column = it[\"column\"]\n    values_X = test_X[column].values\n    test_corr = scipy.stats.pearsonr(\n        values_X,\n        test_Y\n    )[1]\n    it[\"pearsonr_p_value\"] = test_corr\n    \n    perm_column = it[\"perm_column\"]\n    perm_values_X = test_X[perm_column].values\n    perm_test_corr = scipy.stats.pearsonr(\n        perm_values_X,\n        test_Y\n    )[1]\n    it[\"perm_pearsonr_p_value\"] = perm_test_corr","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:51:14.820222Z","iopub.execute_input":"2023-02-28T22:51:14.821538Z","iopub.status.idle":"2023-02-28T22:51:14.845919Z","shell.execute_reply.started":"2023-02-28T22:51:14.821466Z","shell.execute_reply":"2023-02-28T22:51:14.844706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for it in bad_vanil_corrs_columns:\n    column = it[\"column\"]\n    values_X = test_X[column].values\n    test_corr = scipy.stats.spearmanr(\n        values_X,\n        test_Y\n    ).pvalue\n    it[\"spearmanr_p_value\"] = test_corr\n    \n    perm_column = it[\"perm_column\"]\n    perm_values_X = test_X[perm_column].values\n    perm_test_corr = scipy.stats.spearmanr(\n        perm_values_X,\n        test_Y\n    ).pvalue\n    it[\"perm_spearmanr_p_value\"] = perm_test_corr","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:59:08.850398Z","iopub.execute_input":"2023-02-28T22:59:08.851084Z","iopub.status.idle":"2023-02-28T22:59:09.045571Z","shell.execute_reply.started":"2023-02-28T22:59:08.851034Z","shell.execute_reply":"2023-02-28T22:59:09.044165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# bad_vanil_corrs_columns","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:59:11.242699Z","iopub.execute_input":"2023-02-28T22:59:11.243303Z","iopub.status.idle":"2023-02-28T22:59:11.250105Z","shell.execute_reply.started":"2023-02-28T22:59:11.243217Z","shell.execute_reply":"2023-02-28T22:59:11.248522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame(bad_vanil_corrs_columns)","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:59:11.518947Z","iopub.execute_input":"2023-02-28T22:59:11.519551Z","iopub.status.idle":"2023-02-28T22:59:11.529609Z","shell.execute_reply.started":"2023-02-28T22:59:11.519491Z","shell.execute_reply":"2023-02-28T22:59:11.527766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"low_corr_df = df[np.abs(df[\"corr\"]) < 0.1]\nlow_corr_df.sort_values(\"lgbm_corr\", inplace=True, ascending=False)\n\nlow_corr_df[[\n    \"rank\",\n    \"column\",    \n    \"corr\",\n    \"spearmanr\",\n    \"pearsonr_p_value\",\n    \"spearmanr_p_value\",\n    \"lgbm_corr\"\n]]","metadata":{"execution":{"iopub.status.busy":"2023-02-28T23:00:24.091091Z","iopub.execute_input":"2023-02-28T23:00:24.091619Z","iopub.status.idle":"2023-02-28T23:00:24.119017Z","shell.execute_reply.started":"2023-02-28T23:00:24.091575Z","shell.execute_reply":"2023-02-28T23:00:24.117744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_pickle(\"/kaggle/working/fi_perm_single.pkl\")","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:59:13.072742Z","iopub.execute_input":"2023-02-28T22:59:13.073253Z","iopub.status.idle":"2023-02-28T22:59:13.084178Z","shell.execute_reply.started":"2023-02-28T22:59:13.073196Z","shell.execute_reply":"2023-02-28T22:59:13.082601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2023-02-28T22:45:15.527311Z","iopub.execute_input":"2023-02-28T22:45:15.527803Z","iopub.status.idle":"2023-02-28T22:45:15.554283Z","shell.execute_reply.started":"2023-02-28T22:45:15.527761Z","shell.execute_reply":"2023-02-28T22:45:15.552862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}