{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-09-20T10:41:18.341984Z","iopub.execute_input":"2023-09-20T10:41:18.342425Z","iopub.status.idle":"2023-09-20T10:41:18.353822Z","shell.execute_reply.started":"2023-09-20T10:41:18.342391Z","shell.execute_reply":"2023-09-20T10:41:18.352580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Importing libraries","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport datetime\nimport plotly.graph_objs as go\nimport plotly.express as px\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:18.416183Z","iopub.execute_input":"2023-09-20T10:41:18.416586Z","iopub.status.idle":"2023-09-20T10:41:18.423671Z","shell.execute_reply.started":"2023-09-20T10:41:18.416549Z","shell.execute_reply":"2023-09-20T10:41:18.422293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# uploading dataset","metadata":{}},{"cell_type":"code","source":"de_train=pd.read_parquet('/kaggle/input/open-problems-single-cell-perturbations/de_train.parquet')\nde_train","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:18.502241Z","iopub.execute_input":"2023-09-20T10:41:18.502729Z","iopub.status.idle":"2023-09-20T10:41:20.218664Z","shell.execute_reply.started":"2023-09-20T10:41:18.502676Z","shell.execute_reply":"2023-09-20T10:41:20.217775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.shape","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:20.220331Z","iopub.execute_input":"2023-09-20T10:41:20.220973Z","iopub.status.idle":"2023-09-20T10:41:20.230112Z","shell.execute_reply.started":"2023-09-20T10:41:20.220941Z","shell.execute_reply":"2023-09-20T10:41:20.228151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(de_train.columns)\nprint(de_train.index)\nprint(de_train.dtypes)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:20.231659Z","iopub.execute_input":"2023-09-20T10:41:20.232101Z","iopub.status.idle":"2023-09-20T10:41:20.246029Z","shell.execute_reply.started":"2023-09-20T10:41:20.232064Z","shell.execute_reply":"2023-09-20T10:41:20.244795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in de_train.columns:\n    if de_train[i].dtypes=='object':\n        print(i)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:20.249176Z","iopub.execute_input":"2023-09-20T10:41:20.249562Z","iopub.status.idle":"2023-09-20T10:41:21.060454Z","shell.execute_reply.started":"2023-09-20T10:41:20.249529Z","shell.execute_reply":"2023-09-20T10:41:21.059194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:21.062206Z","iopub.execute_input":"2023-09-20T10:41:21.062545Z","iopub.status.idle":"2023-09-20T10:41:21.089773Z","shell.execute_reply.started":"2023-09-20T10:41:21.062516Z","shell.execute_reply":"2023-09-20T10:41:21.088581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:21.091444Z","iopub.execute_input":"2023-09-20T10:41:21.091826Z","iopub.status.idle":"2023-09-20T10:41:21.140612Z","shell.execute_reply.started":"2023-09-20T10:41:21.091795Z","shell.execute_reply":"2023-09-20T10:41:21.139561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(14,4))\nsns.heatmap(de_train.isna())\nplt.title('Clean heatmap of dataset')","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:21.142044Z","iopub.execute_input":"2023-09-20T10:41:21.142372Z","iopub.status.idle":"2023-09-20T10:41:39.614921Z","shell.execute_reply.started":"2023-09-20T10:41:21.142344Z","shell.execute_reply":"2023-09-20T10:41:39.613750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.describe()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:41:39.616432Z","iopub.execute_input":"2023-09-20T10:41:39.616885Z","iopub.status.idle":"2023-09-20T10:42:19.261783Z","shell.execute_reply.started":"2023-09-20T10:41:39.616846Z","shell.execute_reply":"2023-09-20T10:42:19.260349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"de_train.info()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:42:19.263851Z","iopub.execute_input":"2023-09-20T10:42:19.264249Z","iopub.status.idle":"2023-09-20T10:42:19.514839Z","shell.execute_reply.started":"2023-09-20T10:42:19.264207Z","shell.execute_reply":"2023-09-20T10:42:19.513477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization of gene expresssion","metadata":{}},{"cell_type":"code","source":"fig=px.scatter(de_train,x='cell_type',\n               y='sm_name',\n               color='cell_type')\nfig.update_layout(title='Graph of cell type and sm_name',\n                  xaxis_title='CELL TYPE',\n                  yaxis_title='SM NAMES')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:42:19.519222Z","iopub.execute_input":"2023-09-20T10:42:19.519601Z","iopub.status.idle":"2023-09-20T10:42:19.636496Z","shell.execute_reply.started":"2023-09-20T10:42:19.519569Z","shell.execute_reply":"2023-09-20T10:42:19.635138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.style.use('classic')\nfig=px.bar(de_train,x='cell_type',\n           y='sm_name',\n           color='cell_type')\nfig.update_layout(title='Bar Graph of cell type and sm_name',\n                  xaxis_title='CELL TYPE',\n                  yaxis_title='SM NAMES')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:42:19.638119Z","iopub.execute_input":"2023-09-20T10:42:19.638561Z","iopub.status.idle":"2023-09-20T10:42:19.760837Z","shell.execute_reply.started":"2023-09-20T10:42:19.638529Z","shell.execute_reply":"2023-09-20T10:42:19.759559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npair_df=de_train[[ 'A1BG',\n       'A1BG-AS1', 'A2M', 'A2M-AS1', 'A2MP1','ZUP1', 'ZW10', 'ZWILCH', 'ZWINT', 'ZXDA', 'ZXDB', 'ZXDC', 'ZYG11B',\n       'ZYX', 'ZZEF1']]\nprint(pair_df.head())\nsns.pairplot(pair_df)","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:42:19.762452Z","iopub.execute_input":"2023-09-20T10:42:19.762942Z","iopub.status.idle":"2023-09-20T10:44:03.122770Z","shell.execute_reply.started":"2023-09-20T10:42:19.762899Z","shell.execute_reply":"2023-09-20T10:44:03.121788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#distribution plot of  a few columns\npair_df=de_train[[ 'A1BG',\n       'A1BG-AS1', 'A2M', 'A2M-AS1', 'A2MP1','ZUP1']]\nprint(pair_df.head())\nplt.figure(figsize=(14,6))\nsns.displot(pair_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:45:42.386393Z","iopub.execute_input":"2023-09-20T10:45:42.386867Z","iopub.status.idle":"2023-09-20T10:45:52.404139Z","shell.execute_reply.started":"2023-09-20T10:45:42.386821Z","shell.execute_reply":"2023-09-20T10:45:52.402553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\npair_df=de_train[[ 'A1BG',\n       'A1BG-AS1', 'A2M', 'A2M-AS1', 'A2MP1','ZUP1', 'ZW10', 'ZWILCH', 'ZWINT', 'ZXDA', 'ZXDB', 'ZXDC', 'ZYG11B',\n       'ZYX', 'ZZEF1']]\nprint(pair_df.head())\nsns.displot(pair_df)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-09-20T10:44:38.341793Z","iopub.execute_input":"2023-09-20T10:44:38.342376Z","iopub.status.idle":"2023-09-20T10:45:13.925613Z","shell.execute_reply.started":"2023-09-20T10:44:38.342329Z","shell.execute_reply":"2023-09-20T10:45:13.924371Z"},"trusted":true},"execution_count":null,"outputs":[]}]}