{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-11T18:39:34.746664Z","iopub.execute_input":"2023-02-11T18:39:34.747256Z","iopub.status.idle":"2023-02-11T18:39:34.755474Z","shell.execute_reply.started":"2023-02-11T18:39:34.747208Z","shell.execute_reply":"2023-02-11T18:39:34.753836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Key Parameters","metadata":{}},{"cell_type":"code","source":"target_name = 'CD36'#   'CD194' # 'CD44'# 'CD62L'\n\ndown_sample_percent = 50\n\nn_trials_mutual_info_sklearn = 1#  4\n\nlist_n_estimators = [1,2,3,100] # Used for boostings for trial0,trial1, etc... \nn_trials_lgb = 0# 4\nn_trials_catBoost = 0#  4\n\n\nimport time\nt0start = time.time() ","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:39:36.267823Z","iopub.execute_input":"2023-02-11T18:39:36.268266Z","iopub.status.idle":"2023-02-11T18:39:36.275728Z","shell.execute_reply.started":"2023-02-11T18:39:36.268233Z","shell.execute_reply":"2023-02-11T18:39:36.274242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install git+https://github.com/jundongl/scikit-feature","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:39:36.470296Z","iopub.execute_input":"2023-02-11T18:39:36.470695Z","iopub.status.idle":"2023-02-11T18:39:53.489757Z","shell.execute_reply.started":"2023-02-11T18:39:36.470665Z","shell.execute_reply":"2023-02-11T18:39:53.487628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from skfeature.utility.entropy_estimators import midd\nimport numpy as np\nimport time","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:35:08.278035Z","iopub.execute_input":"2023-02-11T18:35:08.278562Z","iopub.status.idle":"2023-02-11T18:35:08.284861Z","shell.execute_reply.started":"2023-02-11T18:35:08.278526Z","shell.execute_reply":"2023-02-11T18:35:08.283684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load Data\n","metadata":{}},{"cell_type":"code","source":"%%time\nfilename_rna_data = '/kaggle/input/open-problems-multimodal/train_cite_inputs.h5'\ndf_rna = pd.read_hdf(filename_rna_data)\ndisplay(df_rna) \n\n#%%time\ndf_y = pd.read_hdf('/kaggle/input/open-problems-multimodal/train_cite_targets.h5')\ndisplay(df_y)\n\nfn = '/kaggle/input/open-problems-multimodal/metadata.csv'\ndf_meta = pd.read_csv(fn, index_col = 0 )\ndf_meta\n# Cut only train cite-seq part: \nd = pd.DataFrame(index = df_y.index)\nprint(d.shape)\ndf_meta = d.join(df_meta, how = 'left')\ndf_meta","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:39:53.493823Z","iopub.execute_input":"2023-02-11T18:39:53.494381Z","iopub.status.idle":"2023-02-11T18:40:57.494369Z","shell.execute_reply.started":"2023-02-11T18:39:53.494336Z","shell.execute_reply":"2023-02-11T18:40:57.492820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print( dict( df_meta.value_counts('cell_type') ))\n","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:47:12.369094Z","iopub.execute_input":"2023-02-11T18:47:12.369511Z","iopub.status.idle":"2023-02-11T18:47:12.385663Z","shell.execute_reply.started":"2023-02-11T18:47:12.369478Z","shell.execute_reply":"2023-02-11T18:47:12.384047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y = df_y[target_name].values\n# print(y.shape, type(y) )\n\nX = ( df_rna.values )\nprint(X.shape)\n\ny = df_y[target_name]\nprint(y.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:47:13.743850Z","iopub.execute_input":"2023-02-11T18:47:13.744233Z","iopub.status.idle":"2023-02-11T18:47:13.751562Z","shell.execute_reply.started":"2023-02-11T18:47:13.744204Z","shell.execute_reply":"2023-02-11T18:47:13.750192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_importances = pd.DataFrame(index = df_rna.columns )","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:47:15.378984Z","iopub.execute_input":"2023-02-11T18:47:15.379398Z","iopub.status.idle":"2023-02-11T18:47:15.386088Z","shell.execute_reply.started":"2023-02-11T18:47:15.379358Z","shell.execute_reply":"2023-02-11T18:47:15.384688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport time\nt0 = time.time()\n\ny = df_y[target_name]\nN = int(down_sample_percent*len(y) / 100 )\nfor trial in range(n_trials_mutual_info_sklearn):\n    l = []\n    for i,col in enumerate( df_rna.columns):\n\n        p = np.random.permutation(len(y))\n\n        x_loc = df_rna[col].values[p][:N]\n        mi = midd(x_loc , y.values[p][:N] )\n        s = mi\n        df_importances.loc[col, target_name +' MutualInfSkfeature DownSample'+str(down_sample_percent) + ' Trial'+str(trial) ]  = s\n\n        l.append(s)\n        if (i%5000==0) or (i<5):\n            I = np.argmax(l)\n            #print(i, col, '%.1f seconds passed'%(time.time() - t0 ), 'Pearson Score', s )\n            print(i, col,'Mean %.6f, max %.6f'%( np.mean(l), np.max(l) ),  \n                  '%.1f seconds passed'%(time.time() - t0 ), 'Argmax:', I, df_rna.columns[I] )            \n        \n    \ndf_importances","metadata":{"execution":{"iopub.status.busy":"2023-02-11T18:47:43.981002Z","iopub.execute_input":"2023-02-11T18:47:43.981481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}