{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  Transform the input dataset by column's prefix\n\nWith respect to the idea of \"Multiome Quickstart\": \nhttps://www.kaggle.com/code/ambrosm/msci-multiome-quickstart/notebook \n I suggest to transform the original dataset by grouping columns with similar prefixes, which is a part of GENE_ID, such as 'chr9', 'chrX' etc.\n\n> cols_letter_lst = list(set([x[:CATEGOR_LETTERS] for x in cols[:]]))\n\nCurrent approach addresses the issue of feature extraction from the dataset with 'too many' columns, making the input more sutable for the classification model and critical for cases of limited RAM.\n\nThis version adds more granularity and makes the input more sensitive (crispy).\n","metadata":{}},{"cell_type":"code","source":"## With referemces to other notebooks:\n##  https://www.kaggle.com/code/ambrosm/msci-multiome-quickstart/notebook\n\nimport os, gc, pickle\nimport pandas as pd\nimport numpy as np\nimport datetime\n\nfrom sklearn.linear_model import Lasso\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom colorama import Fore, Back, Style\n\nDATA_DIR = \"/kaggle/input/open-problems-multimodal/\"\nFP_MULTIOME_TRAIN_INPUTS = os.path.join(DATA_DIR,\"train_multi_inputs.h5\")\n\n# We need this library to read HDF files.\n!pip install --quiet tables\nprint(datetime.datetime.today())\nprint(\"\")\n\n## Read the sample of the input data\nprint(f\"{Fore.BLACK}{Style.BRIGHT}  Read sample of the input data {Style.RESET_ALL}\")\ntrain_input_df = pd.read_hdf(FP_MULTIOME_TRAIN_INPUTS, start=0, stop=10)\ncols = train_input_df.columns.tolist()\nprint(\"Total columns in the original data: \", len(cols))\n\n## Find columns by prefix\nCATEGOR_LETTERS = 5 \ncol_subnames_lst = list(set([x[:CATEGOR_LETTERS] for x in cols[:]]))\nprint(\"Number of sub-names: \", len(col_subnames_lst))\n\nprint(\"Example of grouping columns by prefix\")\ncol_stats = []\nfor col_name in col_subnames_lst[0:5]:\n    cols = train_input_df.filter(regex='^{0}'.format(col_name)).columns\n    col_stats_dict = {}\n    col_stats_dict['col_subname'] = col_name\n    col_stats_dict['col_count'] = len(cols)\n    col_stats.append(col_stats_dict)\nstats_df = pd.DataFrame(col_stats)\ndisplay(stats_df)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, let's develop a function that transforms all input columns by:\n* 1. Grouping sub-names of the feature\n* 2. Extracting features for each group of columns","metadata":{}},{"cell_type":"code","source":"## \n## Grouping coulmns by prefix and extracting features.\n## The original dataset has thousands of columns,\n## after tranformation - will be preseted as 200-300 column-features\n##\ndef transform_multiome_input_columns(input_df, DBG=True):\n    \n    print(\"FP_MULTIOME_TRAIN_INPUTS\")\n    print(\"Total original columns: \", len(input_df.columns))\n\n    cols = input_df.columns.tolist()\n    CATEGOR_LETTERS = 5 \n    cols_letter_lst = list(set([x[:CATEGOR_LETTERS] for x in cols[:]]))\n    print(\"cols_letter_lst: \", cols_letter_lst)\n\n    COL_INDEX = input_df.index\n    COL_NAMES = []\n    COL_STATS = []\n    \n    for col_name in cols_letter_lst:\n        \n        cols = input_df.filter(regex='^{0}'.format(col_name)).columns\n        col_stats_dict = {}\n        col_stats_dict['col_name'] = col_name\n        col_stats_dict['col_count'] = len(cols)\n        \n        ## New column names per iterration for debugging\n        NEW_NAMES = []\n        total_cols = len(cols)\n        \n        ## Stats\n        input_df['mean_{0}'.format(col_name)] = input_df[cols].astype('float').mean(axis=1)\n        input_df['max_{0}'.format(col_name)] = input_df[cols].astype('float').max(axis=1)\n        COL_NAMES.append('mean_{0}'.format(col_name))\n        COL_NAMES.append('max_{0}'.format(col_name))\n        NEW_NAMES.append('mean_{0}'.format(col_name))\n        NEW_NAMES.append('max_{0}'.format(col_name))\n        \n        ## Datapoints above thresholds\n        G_LST = [0.5, 1, 2, 2.5, 3, 3.5, 4, 4.5]\n        for idx_g in G_LST:\n            th = idx_g\n            count_positive_idx = input_df[cols].gt(th, axis=1).sum(axis=1)\n            input_df['count_gt_{0}_{1}'.format(th, col_name)] = (count_positive_idx/total_cols).round(6)\n            COL_NAMES.append('count_gt_{0}_{1}'.format(th, col_name))\n            NEW_NAMES.append('count_gt_{0}_{1}'.format(th, col_name))\n        \n        ## Datapoints in buckets\n        b1 = [0.5, 1, 2.5, 3.5, 4, 4.5]\n        b2 = [2,   3, 4,   5,   6, 7]\n        for idx in range(0,len(b1)):\n            th1 = b1[idx] \n            th2 = b2[idx] \n            count_bucket_idx = input_df[cols][input_df[cols]>th1].lt(th2).sum(axis=1)/total_cols\n            input_df['count_bucket_{0}_{1}'.format(th1, col_name)] = (count_bucket_idx).round(6)\n            COL_NAMES.append('count_bucket_{0}_{1}'.format(th1, col_name))\n            NEW_NAMES.append('count_bucket_{0}_{1}'.format(th1, col_name))\n\n        COL_STATS.append(col_stats_dict)\n        \n        if DBG:\n            print(\"Features for '{0}'\". format(col_name))\n            display(input_df[NEW_NAMES].head(4))\n        \n    input_df.fillna(0)\n    \n    return input_df[COL_NAMES], COL_STATS\n\n\nprint(datetime.datetime.today())\n\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we invoke the *transformation function* by reading the original dataset in chunks. \nOnce we collect enough data in a compact format, we save it in one file for further processing.","metadata":{}},{"cell_type":"code","source":"print(datetime.datetime.today())\n\n## Read data in chunks and collected tranformed data in a compact format\nCHUNK_SIZE = 5000\nCHUNKS_READ = 4\nDBG_ = True\nmulti_train_x = []\nfor i in range(CHUNKS_READ):\n    \n    r_start, r_stop = i * CHUNK_SIZE, (i+1) * CHUNK_SIZE\n        \n    train_input_df = pd.read_hdf(FP_MULTIOME_TRAIN_INPUTS, start=r_start, stop=r_stop)\n    print(\"Process FP_MULTIOME_TRAIN_INPUTS\")\n\n    ##########################################################\n    print(\"--------------------------------------------------\")\n    trans_df, COL_STATS = transform_multiome_input_columns(train_input_df, DBG_)\n    ##########################################################\n    \n    print(datetime.datetime.today())\n    \n    if len(multi_train_x) == 0:\n        multi_train_x = trans_df.copy()\n        DBG_ = False\n    else:\n        multi_train_x = np.concatenate([multi_train_x, trans_df.copy()])\n            \n    print(\"Collected multi_train_x: \", multi_train_x.shape)\n\n    del train_input_df, trans_df\n\nd = gc.collect()\n\nprint(\"Total records for training: \", multi_train_x.shape)\ntrain_rows = multi_train_x.shape[0]\n\nprint(\"********************************************************\")\nprint(\"Training input size after column tranformation : \",  multi_train_x.shape)\n\nprint(\"Saving multi_train_x for further processing\")\nwith open(\"multi_train_x.pickle\", 'wb') as f: pickle.dump(multi_train_x, f)\n\ndel multi_train_x","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Read input data and fit the model\nimport warnings\nwarnings.filterwarnings('ignore')\n\nprint(\"Read transformed training file \", datetime.datetime.today())\nwith open(\"/kaggle/working/multi_train_x.pickle\", 'rb') as f: multi_train_x = pickle.load(f)\ntrain_rows = multi_train_x.shape[0]\nFP_MULTIOME_TRAIN_TARGETS = os.path.join(DATA_DIR,\"train_multi_targets.h5\")\nmulti_train_y_df = pd.read_hdf(FP_MULTIOME_TRAIN_TARGETS, start=0, stop=train_rows)\n\nprint(\"        Training INPUT:  \", multi_train_x.shape)\nprint(\"********************************************************\")\nprint(\"        Training TARGET: \", multi_train_y_df.shape)\n\nprint(\"Training the model ..\")\nprint(datetime.datetime.today())\nprint(\"------------------------\")\n      \nLASSO_ITER = 50 ## A quick version\nALP = 0.2\nmodel = Lasso(copy_X=False, max_iter= LASSO_ITER, alpha=ALP)\n## Train the model\nprint(\"Training the model .. It might take an hour\")\nmodel.fit(multi_train_x, multi_train_y_df)\n\nprint(\"The model has been trained\")    \ndel multi_train_x, multi_train_y_df # free the RAM\nd = gc.collect()\n\nprint(datetime.datetime.today())\nprint(\"------------------------\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"After the model got trained, it is ready for predicton or can be temorary saved into a picle file.","metadata":{}}]}