{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-28T10:59:43.491175Z","iopub.execute_input":"2022-08-28T10:59:43.493695Z","iopub.status.idle":"2022-08-28T10:59:43.504530Z","shell.execute_reply.started":"2022-08-28T10:59:43.493624Z","shell.execute_reply":"2022-08-28T10:59:43.502954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# required to sklearn tsne\n!pip install tables","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:59:43.507461Z","iopub.execute_input":"2022-08-28T10:59:43.508490Z","iopub.status.idle":"2022-08-28T10:59:58.472468Z","shell.execute_reply.started":"2022-08-28T10:59:43.508440Z","shell.execute_reply":"2022-08-28T10:59:58.470568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA: load metadata","metadata":{}},{"cell_type":"code","source":"meta = pd.read_csv(\"/kaggle/input/open-problems-multimodal/metadata.csv\")\n# set index as cell_id\nmeta.set_index('cell_id', inplace=True)\nmeta.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:59:58.475127Z","iopub.execute_input":"2022-08-28T10:59:58.475911Z","iopub.status.idle":"2022-08-28T10:59:58.965419Z","shell.execute_reply.started":"2022-08-28T10:59:58.475862Z","shell.execute_reply":"2022-08-28T10:59:58.964335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# show distribution by cell type and technology\nmeta.hist(column=\"cell_type\", by=\"technology\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:59:58.966844Z","iopub.execute_input":"2022-08-28T10:59:58.967969Z","iopub.status.idle":"2022-08-28T10:59:59.587933Z","shell.execute_reply.started":"2022-08-28T10:59:58.967929Z","shell.execute_reply":"2022-08-28T10:59:59.586624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remove features / transcripts with low variance.\n\nTODO: check if the dataset was already normalized.","metadata":{}},{"cell_type":"code","source":"# load transcriptomics information\ntrain_cite_inputs = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_inputs.h5\",\n    stop=1000 # need to limit due to memory constrains\n    #iterator=True,\n    #chunksize=5000\n)\nprint(f\"train_cite_inputs dim: {train_cite_inputs.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T10:59:59.591565Z","iopub.execute_input":"2022-08-28T10:59:59.592088Z","iopub.status.idle":"2022-08-28T11:00:00.509085Z","shell.execute_reply.started":"2022-08-28T10:59:59.592033Z","shell.execute_reply":"2022-08-28T11:00:00.507564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Remove features with low variance","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_selection import VarianceThreshold\n\ntrain_cite_fltr = VarianceThreshold(threshold=0.2).fit_transform(train_cite_inputs)\nprint(f\"train_cite_fltr dim: {train_cite_fltr.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:00:00.510809Z","iopub.execute_input":"2022-08-28T11:00:00.511159Z","iopub.status.idle":"2022-08-28T11:00:02.170310Z","shell.execute_reply.started":"2022-08-28T11:00:00.511128Z","shell.execute_reply":"2022-08-28T11:00:02.168656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"generate TSNE","metadata":{}},{"cell_type":"code","source":"from sklearn.manifold import TSNE\n# set components\nn_components=2\n# make component col names\ncol_names = [ f\"tsne-2d-{str(i)}\" for i in range(1, n_components + 1) ]\n# create TSNE\ntsne_cite = TSNE(\n    n_components=n_components, \n    learning_rate='auto', \n    init='random', \n    perplexity=3\n)\n# fit TSNE\ntsne_fit = tsne_cite.fit_transform(train_cite_fltr)\n# convert to df\nembedded = pd.DataFrame(\n    tsne_fit, \n    columns = col_names)\n# add index as cell id\nembedded.set_index(train_cite_inputs.index, inplace=True)\nembedded.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:32:25.103902Z","iopub.execute_input":"2022-08-28T11:32:25.105274Z","iopub.status.idle":"2022-08-28T11:32:30.043603Z","shell.execute_reply.started":"2022-08-28T11:32:25.105203Z","shell.execute_reply":"2022-08-28T11:32:30.042481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"concatenate to metadata","metadata":{}},{"cell_type":"code","source":"# combine embedded to metadata\ntsne_table = pd.concat([embedded, meta], axis=1, join=\"inner\")\ntsne_table.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:00:07.171402Z","iopub.execute_input":"2022-08-28T11:00:07.171768Z","iopub.status.idle":"2022-08-28T11:00:07.313205Z","shell.execute_reply.started":"2022-08-28T11:00:07.171736Z","shell.execute_reply":"2022-08-28T11:00:07.312287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"generate figures TSNE","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# set variable for hue\ncolor_by = 'cell_type'\n# set plot size\nplt.figure(figsize=(16,10))\n# generate scatterplot\nsns.scatterplot(\n    x=\"tsne-2d-1\", y=\"tsne-2d-2\",\n    hue=color_by,\n    palette=sns.color_palette(\"hls\", len(tsne_table[color_by].unique())),\n    data=tsne_table,\n    legend=\"full\",\n    alpha=0.3\n)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:00:07.314800Z","iopub.execute_input":"2022-08-28T11:00:07.315855Z","iopub.status.idle":"2022-08-28T11:00:08.082461Z","shell.execute_reply.started":"2022-08-28T11:00:07.315815Z","shell.execute_reply":"2022-08-28T11:00:08.081109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Train Predictor for cell type class.","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\nfrom sklearn.tree import DecisionTreeClassifier\n\nclf = DecisionTreeClassifier(random_state=0)\ncross_val_score(clf, tsne_table[[\"tsne-2d-1\", \"tsne-2d-2\"]], tsne_table.cell_type, cv=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:00:08.084338Z","iopub.execute_input":"2022-08-28T11:00:08.085199Z","iopub.status.idle":"2022-08-28T11:00:08.186004Z","shell.execute_reply.started":"2022-08-28T11:00:08.085143Z","shell.execute_reply":"2022-08-28T11:00:08.184601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# check /kaggle/input/open-problems-multimodal/train_multi_inputs.h5\n","metadata":{}},{"cell_type":"code","source":"# load transcriptomics information\ntrain_cite_targets = pd.read_hdf(\n    \"/kaggle/input/open-problems-multimodal/train_cite_targets.h5\",\n    stop=1000 # need to limit due to memory constrains\n    #iterator=True,\n    #chunksize=5000\n)\nprint(f\"train_cite_targets dim: {train_cite_targets.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:01:20.028731Z","iopub.execute_input":"2022-08-28T11:01:20.029203Z","iopub.status.idle":"2022-08-28T11:01:20.074399Z","shell.execute_reply.started":"2022-08-28T11:01:20.029167Z","shell.execute_reply":"2022-08-28T11:01:20.073156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.manifold import TSNE\n# make component col names\ncol_names = [ f\"tsne-2d-target-{str(i)}\" for i in range(1, n_components + 1) ]\n# fit TSNE\ntsne_fit = tsne_cite.fit_transform(train_cite_targets)\n# convert to df\nembedded = pd.DataFrame(\n    tsne_fit, \n    columns = col_names)\n# add index as cell id\nembedded.set_index(train_cite_targets.index, inplace=True)\nembedded.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:15:03.710700Z","iopub.execute_input":"2022-08-28T11:15:03.711083Z","iopub.status.idle":"2022-08-28T11:15:07.700660Z","shell.execute_reply.started":"2022-08-28T11:15:03.711052Z","shell.execute_reply":"2022-08-28T11:15:07.699647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# combine embedded to metadata\ntsne_table_targets = pd.concat([embedded, tsne_table], axis=1, join=\"inner\")\ntsne_table_targets.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:15:11.608480Z","iopub.execute_input":"2022-08-28T11:15:11.608922Z","iopub.status.idle":"2022-08-28T11:15:11.627840Z","shell.execute_reply.started":"2022-08-28T11:15:11.608889Z","shell.execute_reply":"2022-08-28T11:15:11.626435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\n# set variable for hue\ncolor_by = 'cell_type'\n# set plot size\nplt.figure(figsize=(16,10))\n# generate scatterplot\nsns.scatterplot(\n    x=\"tsne-2d-target-1\", y=\"tsne-2d-target-2\",\n    hue=color_by,\n    #palette=sns.color_palette(\"hls\", len(col_names)),\n    data=tsne_table_targets,\n    legend=\"full\",\n    alpha=0.3\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:16:35.324664Z","iopub.execute_input":"2022-08-28T11:16:35.325070Z","iopub.status.idle":"2022-08-28T11:16:36.118529Z","shell.execute_reply.started":"2022-08-28T11:16:35.325036Z","shell.execute_reply":"2022-08-28T11:16:36.117247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import MultiTaskElasticNet\nfrom sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(train_cite_inputs, train_cite_targets, random_state=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:38:34.863605Z","iopub.execute_input":"2022-08-28T11:38:34.864032Z","iopub.status.idle":"2022-08-28T11:38:34.995757Z","shell.execute_reply.started":"2022-08-28T11:38:34.864001Z","shell.execute_reply":"2022-08-28T11:38:34.994785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclf = MultiTaskElasticNet(alpha=0.1)\n\nclf.fit(X=X_train, y=y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:38:36.059339Z","iopub.execute_input":"2022-08-28T11:38:36.060121Z","iopub.status.idle":"2022-08-28T11:46:51.558825Z","shell.execute_reply.started":"2022-08-28T11:38:36.060083Z","shell.execute_reply":"2022-08-28T11:46:51.552054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf.score(X_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:51:59.136345Z","iopub.execute_input":"2022-08-28T11:51:59.136761Z","iopub.status.idle":"2022-08-28T11:51:59.362388Z","shell.execute_reply.started":"2022-08-28T11:51:59.136728Z","shell.execute_reply":"2022-08-28T11:51:59.360808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-28T11:36:39.535082Z","iopub.execute_input":"2022-08-28T11:36:39.535474Z","iopub.status.idle":"2022-08-28T11:36:39.572249Z","shell.execute_reply.started":"2022-08-28T11:36:39.535442Z","shell.execute_reply":"2022-08-28T11:36:39.570926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}