{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-08T03:36:20.527223Z","iopub.execute_input":"2022-11-08T03:36:20.527559Z","iopub.status.idle":"2022-11-08T03:36:20.539889Z","shell.execute_reply.started":"2022-11-08T03:36:20.527532Z","shell.execute_reply":"2022-11-08T03:36:20.538513Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#In this notebook I would like to investigate RNA marker correlations further. \n#I looked at all RNA asssociations of CITE/Protein Cluster Differentiation(CD) target markers in the training dataset.\n#I found after looking at the genes associated with the CITE-protein targets that some \n#genes such as:'HLA-A-B-C','CD105','CD56','CD152','CD2','CD29','CD134','CD1c','CD48','HLA-DR', had a higher degree of correlation with other genes.\n#It may be better to use these gene associations for prediction models.\n#CD152 showed a better degree of cd152-RNA and cd152-protein association in the training dataset\n#with a pair-wise correlation scrore of 0.148\n# I tried Linear regression, however data appears to follow a logistic/binomial distribution trend which is expected of scRNA-seq data sets\n#CD134-RNA and HLR-DR-RNA has a pirwise correlation scrore of 0.25","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:36:20.541824Z","iopub.execute_input":"2022-11-08T03:36:20.542191Z","iopub.status.idle":"2022-11-08T03:36:20.547517Z","shell.execute_reply.started":"2022-11-08T03:36:20.542157Z","shell.execute_reply":"2022-11-08T03:36:20.546103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from matplotlib.colors import ListedColormap, LinearSegmentedColormap\nfrom matplotlib.offsetbox import AnnotationBbox, OffsetImage\nfrom sklearn.decomposition import TruncatedSVD\nfrom matplotlib.patches import Rectangle\nimport matplotlib.patches as patches\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm\nimport seaborn as sns\nimport pandas as pd\nimport numpy as np\nimport itertools\nimport warnings\nimport os\nimport gc\n!pip install scanpy\nimport scanpy as sc\nimport anndata\nimport time\nt0start = time.time()\nimport sys","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:36:20.560308Z","iopub.execute_input":"2022-11-08T03:36:20.560646Z","iopub.status.idle":"2022-11-08T03:36:35.905557Z","shell.execute_reply.started":"2022-11-08T03:36:20.560618Z","shell.execute_reply":"2022-11-08T03:36:35.903914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAIN_DIRECT = '../input/open-problems-multimodal'\nmeta_data_path = os.path.join(MAIN_DIRECT + '/metadata.csv')\neval_data_path  = os.path.join(MAIN_DIRECT + '/evaluation_ids.csv')\nsubmission_path = os.path.join(MAIN_DIRECT + '/sample_submission.csv')\n\ntest_cite_path = os.path.join(MAIN_DIRECT + '/test_cite_inputs.h5')\ntest_cite_path_day2 = os.path.join(MAIN_DIRECT + '/test_cite_inputs_day_2_donor_27678.h5')\ntest_multi_path = os.path.join(MAIN_DIRECT + '/test_multi_inputs.h5')\n\n\ntrain_cite_path = os.path.join(MAIN_DIRECT + '/train_cite_inputs.h5')\ntrain_cite_target_path = os.path.join(MAIN_DIRECT + '/train_cite_targets.h5')\ntrain_multi_path = os.path.join(MAIN_DIRECT + '/train_multi_inputs.h5')\ntrain_multi_target_path = os.path.join(MAIN_DIRECT + '/train_multi_targets.h5')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:36:35.916468Z","iopub.execute_input":"2022-11-08T03:36:35.916833Z","iopub.status.idle":"2022-11-08T03:36:35.935143Z","shell.execute_reply.started":"2022-11-08T03:36:35.916805Z","shell.execute_reply":"2022-11-08T03:36:35.933458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf_cite_train_x = pd.read_hdf(train_cite_path)\ndf_cite_train_y = pd.read_hdf(train_cite_target_path)\n\nif 0:\n    df_cite_test_x = pd.read_hdf(test_cite_path)\n    display( df_cite_test_x.head() )","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:36:35.938268Z","iopub.execute_input":"2022-11-08T03:36:35.938717Z","iopub.status.idle":"2022-11-08T03:37:15.501112Z","shell.execute_reply.started":"2022-11-08T03:36:35.938681Z","shell.execute_reply":"2022-11-08T03:37:15.500193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#There was a notebook that did help with getting the data in the following table below so I could \n# make my plots and inputed my genes of interest (from lineage markers). I did not note the notebook, \n#but would like to give them credit.\n# (please let me know so I can aknowledge you- Thanks!)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:15.502127Z","iopub.execute_input":"2022-11-08T03:37:15.502361Z","iopub.status.idle":"2022-11-08T03:37:15.507265Z","shell.execute_reply.started":"2022-11-08T03:37:15.502339Z","shell.execute_reply":"2022-11-08T03:37:15.506173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nlist_genes_names = [t.split('_')[1] for t in df_cite_train_x.columns ]\nprint(len(list_genes_names), list_genes_names[:10])\nlist_genes_ids = [t.split('_')[0] for t in df_cite_train_x.columns ]\nprint(len(list_genes_ids), list_genes_ids[:10])","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:15.508913Z","iopub.execute_input":"2022-11-08T03:37:15.509268Z","iopub.status.idle":"2022-11-08T03:37:15.534655Z","shell.execute_reply.started":"2022-11-08T03:37:15.509237Z","shell.execute_reply":"2022-11-08T03:37:15.533111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'ENSG00000066044' in list_genes_ids","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:15.535899Z","iopub.execute_input":"2022-11-08T03:37:15.536191Z","iopub.status.idle":"2022-11-08T03:37:15.544613Z","shell.execute_reply.started":"2022-11-08T03:37:15.536166Z","shell.execute_reply":"2022-11-08T03:37:15.543462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndf_new = pd.DataFrame()\ndf_new['CD3_protein'] = df_cite_train_y['CD3']\n\nlist_ids = ['ENSG00000121410',\n'ENSG00000268895',\n'ENSG00000175899',\n'ENSG00000245105',\n'ENSG00000166535',\n'ENSG00000128274',\n'ENSG00000094914',\n'ENSG00000081760',\n'ENSG00000109576',\n'ENSG00000103591',\n'ENSG00000115977',\n'ENSG00000087884',\n'ENSG00000127837',\n'ENSG00000129673',\n'ENSG00000131043',\n'ENSG00000205002',\n'ENSG00000090861',\n'ENSG00000124608',\n'ENSG00000266967',\n'ENSG00000157426',\n'ENSG00000149313',\n'ENSG00000008311',\n'ENSG00000215458',\n'ENSG00000275700',\n'ENSG00000181409',\n'ENSG00000281376',\n'ENSG00000183044',\n'ENSG00000165029',\n'ENSG00000154263',\n'ENSG00000251595',\n'ENSG00000179869',\n'ENSG00000154262',\n'ENSG00000064687',\n'ENSG00000085563',\n'ENSG00000135776',\n'ENSG00000073734',\n'ENSG00000115657',\n'ENSG00000131269',\n'ENSG00000197150',\n'ENSG00000150967',\n'ENSG00000103222',\n'ENSG00000124574',\n'ENSG00000243064',\n'ENSG00000023839',\n'ENSG00000108846',\n'ENSG00000125257',\n'ENSG00000114770',\n'ENSG00000091262',\n'ENSG00000101986',\n'ENSG00000117528',\n'ENSG00000119688',\n'ENSG00000164163',\n'ENSG00000204574',\n'ENSG00000033050',\n'ENSG00000161204',\n'ENSG00000160179',\n'ENSG00000118777',\n'ENSG00000143921',\n'ENSG00000143994',\n'ENSG00000144827',\n'ENSG00000106077',\n'ENSG00000225969',\n'ENSG00000100997',\n'ENSG00000131969',\n'ENSG00000139826',\n'ENSG00000248487',\n'ENSG00000114786',\n'ENSG00000114779',\n'ENSG00000168792',\n'ENSG00000264031',\n'ENSG00000204427',\n'ENSG00000129968',\n'ENSG00000250536',\n'ENSG00000107362',\n'ENSG00000136379',\n'ENSG00000164074',\n'ENSG00000140526',\n'ENSG00000158201',\n'ENSG00000100439',\n'ENSG00000127220',\n'ENSG00000136754',\n'ENSG00000138443',\n'ENSG00000108798',\n'ENSG00000154175',\n'ENSG00000097007',\n'ENSG00000143322',\n'ENSG00000099204',\n'ENSG00000173210',\n'ENSG00000175164',\n'ENSG00000159842',\n'ENSG00000146386',\n'ENSG00000163322',\n'ENSG00000165660',\n'ENSG00000146109',\n'ENSG00000166016',\n'ENSG00000260246',\n'ENSG00000185065',\n'ENSG00000273212',\n'ENSG00000273300',\n'ENSG00000235776',\n'ENSG00000243107',\n'ENSG00000279265',\n'ENSG00000280347',\n'ENSG00000278727',\n'ENSG00000274898',\n'ENSG00000283208',\n'ENSG00000272330',\n'ENSG00000279591',\n'ENSG00000213683',\n'ENSG00000273216',\n'ENSG00000248636',\n'ENSG00000237729',\n'ENSG00000265975',\n'ENSG00000266389',\n'ENSG00000267698',\n'ENSG00000268947',\n'ENSG00000235560',\n'ENSG00000239791',\n'ENSG00000278922',\n'ENSG00000261433',\n'ENSG00000278993',\n'ENSG00000232392',\n'ENSG00000283638',\n'ENSG00000237819',\n'ENSG00000223969',\n'ENSG00000260188',\n'ENSG00000241764',\n'ENSG00000272829',\n'ENSG00000284060',\n'ENSG00000235945']\nlist_names = ['CD86',\n'CD274',\n'CD270',\n'CD155',\n'CD112',\n'CD47',\n'CD48',\n'CD40',\n'CD154',\n'CD52',\n'CD3',\n'CD8',\n'CD56',\n'CD19',\n'CD33',\n'CD11c',\n'HLA-A-B-C',\n'CD45RA',\n'CD123',\n'CD7',\n'CD105',\n'CD49f',\n'CD194',\n'CD4',\n'CD44',\n'CD14',\n'CD16',\n'CD25',\n'CD45RO',\n'CD279',\n'TIGIT',\n'CD20',\n'CD335',\n'CD31',\n'Podoplanin',\n'CD146',\n'CD5',\n'CD195',\n'CD32',\n'CD196',\n'CD185',\n'CD103',\n'CD69',\n'CD62L',\n'CD161',\n'CD152',\n'CD223',\n'KLRG1',\n'CD27',\n'CD107a',\n'CD95',\n'CD134',\n'HLA-DR',\n'CD1c',\n'CD11b',\n'CD64',\n'CD141',\n'CD1d',\n'CD314',\n'CD35',\n'CD57',\n'CD272',\n'CD278',\n'CD58',\n'CD39',\n'CX3CR1',\n'CD24',\n'CD21',\n'CD11a',\n'CD79b',\n'CD244',\n'CD169',\n'integrinB7',\n'CD268',\n'CD42b',\n'CD54',\n'CD62P',\n'CD119',\n'TCR',\n'CD192',\n'CD122',\n'FceRIa',\n'CD41',\n'CD137',\n'CD163',\n'CD83',\n'CD124',\n'CD13',\n'CD2',\n'CD226',\n'CD29',\n'CD303',\n'CD49b',\n'CD81',\n'CD18',\n'CD28',\n'CD38',\n'CD127',\n'CD45',\n'CD22',\n'CD71',\n'CD26',\n'CD115',\n'CD63',\n'CD304',\n'CD36',\n'CD172a',\n'CD72',\n'CD158',\n'CD93',\n'CD49a',\n'CD49d',\n'CD73',\n'CD9',\n'LOX-1',\n'CD158b',\n'CD158e1',\n'CD142',\n'CD319',\n'CD352',\n'CD94',\n'CD162',\n'CD85j',\n'CD23',\n'CD328',\n'HLA-E',\n'CD82',\n'CD101',\n'CD88',\n'CD224',]\nfor i,id1 in enumerate(list_ids): \n    gene_name1 = list_names[i]\n    IX = list_genes_ids.index(id1)\n    print(IX, list_genes_ids[IX], list_genes_names[IX], df_cite_train_x.columns[IX] )\n    v =  df_cite_train_x.iloc[:, IX]\n    df_new[gene_name1 + '_RNA'] = v\n\n\ndisplay(df_new)\ndisplay(df_new.describe() )\ndisplay(df_new.corr() )","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:15.546691Z","iopub.execute_input":"2022-11-08T03:37:15.547098Z","iopub.status.idle":"2022-11-08T03:37:19.068466Z","shell.execute_reply.started":"2022-11-08T03:37:15.547063Z","shell.execute_reply":"2022-11-08T03:37:19.067258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df_new.corr()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:19.069882Z","iopub.execute_input":"2022-11-08T03:37:19.070384Z","iopub.status.idle":"2022-11-08T03:37:21.206081Z","shell.execute_reply.started":"2022-11-08T03:37:19.070356Z","shell.execute_reply":"2022-11-08T03:37:21.204940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:21.209415Z","iopub.execute_input":"2022-11-08T03:37:21.209702Z","iopub.status.idle":"2022-11-08T03:37:21.224899Z","shell.execute_reply.started":"2022-11-08T03:37:21.209677Z","shell.execute_reply":"2022-11-08T03:37:21.223662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('raw_data.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:21.226759Z","iopub.execute_input":"2022-11-08T03:37:21.227354Z","iopub.status.idle":"2022-11-08T03:37:21.263086Z","shell.execute_reply.started":"2022-11-08T03:37:21.227319Z","shell.execute_reply":"2022-11-08T03:37:21.261882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#I then looked into this 'raw' dataset csv with the marker matrix correlation and \n#pulled out markers that correlated to a higher level as the \n#SNS plot below has to many markers to identify","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:21.264429Z","iopub.execute_input":"2022-11-08T03:37:21.264761Z","iopub.status.idle":"2022-11-08T03:37:21.269734Z","shell.execute_reply.started":"2022-11-08T03:37:21.264730Z","shell.execute_reply":"2022-11-08T03:37:21.268539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport seaborn as sns\nimport matplotlib.pyplot as plt\ncorr = df_new.corr()\nsns.heatmap(corr, \n            xticklabels=corr.columns.values,\n            yticklabels=corr.columns.values)\nplt.show()\nplt.savefig('books_read.png')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:21.271219Z","iopub.execute_input":"2022-11-08T03:37:21.271592Z","iopub.status.idle":"2022-11-08T03:37:26.018677Z","shell.execute_reply.started":"2022-11-08T03:37:21.271558Z","shell.execute_reply":"2022-11-08T03:37:26.017226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndf_new = pd.DataFrame()\ndf_new['CD152_protein'] = df_cite_train_y['CD152']\n\nlist_ids = ['ENSG00000090861',\n'ENSG00000149313',\n'ENSG00000127837',\n'ENSG00000087884',\n'ENSG00000275700',\n'ENSG00000125257',\n'ENSG00000175164',\n'ENSG00000146386',\n'ENSG00000164163',\n'ENSG00000033050',\n'ENSG00000094914',\n'ENSG00000204574']\nlist_names = ['HLA-A-B-C',\n'CD105',\n'CD56',\n'CD8',\n'CD4',\n'CD152',\n'CD2',\n'CD29',\n'CD134',\n'CD1c',\n'CD48',\n'HLA-DR']\nfor i,id1 in enumerate(list_ids): \n    gene_name1 = list_names[i]\n    IX = list_genes_ids.index(id1)\n    print(IX, list_genes_ids[IX], list_genes_names[IX], df_cite_train_x.columns[IX] )\n    v =  df_cite_train_x.iloc[:, IX]\n    df_new[gene_name1 + '_RNA'] = v\n\ndisplay(df_new)\ndisplay(df_new.describe() )\ndisplay(df_new.corr() )","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:00:05.582790Z","iopub.execute_input":"2022-11-08T04:00:05.583239Z","iopub.status.idle":"2022-11-08T04:00:05.798510Z","shell.execute_reply.started":"2022-11-08T04:00:05.583202Z","shell.execute_reply":"2022-11-08T04:00:05.797088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nimport seaborn as sns\nimport matplotlib.pyplot as plt\ncorr = df_new.corr()\nsns.heatmap(corr, \n            xticklabels=corr.columns.values,\n            yticklabels=corr.columns.values)\nplt.show()\nplt.savefig('books_read.png')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T04:00:22.667880Z","iopub.execute_input":"2022-11-08T04:00:22.668314Z","iopub.status.idle":"2022-11-08T04:00:22.974773Z","shell.execute_reply.started":"2022-11-08T04:00:22.668282Z","shell.execute_reply":"2022-11-08T04:00:22.973893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\ndf6 = pd.merge(df_cite_train_x, df_cite_train_y, on='cell_id')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:26.602195Z","iopub.execute_input":"2022-11-08T03:37:26.602649Z","iopub.status.idle":"2022-11-08T03:37:32.754616Z","shell.execute_reply.started":"2022-11-08T03:37:26.602620Z","shell.execute_reply":"2022-11-08T03:37:32.752637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(df6)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:32.756624Z","iopub.execute_input":"2022-11-08T03:37:32.756976Z","iopub.status.idle":"2022-11-08T03:37:32.813318Z","shell.execute_reply.started":"2022-11-08T03:37:32.756942Z","shell.execute_reply":"2022-11-08T03:37:32.811962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn import svm \nimport seaborn as sns \nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import mean_squared_error\nimport numpy as np\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import datasets\nfrom sklearn import svm\nimport pandas as pd\nfrom sklearn import preprocessing, svm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LinearRegression\n","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:44:03.468691Z","iopub.execute_input":"2022-11-08T03:44:03.469114Z","iopub.status.idle":"2022-11-08T03:44:03.476357Z","shell.execute_reply.started":"2022-11-08T03:44:03.469083Z","shell.execute_reply":"2022-11-08T03:44:03.475221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ENSG00000125257_ABCC4_CD152 = df6[[\"ENSG00000125257_ABCC4\", \"CD152\"]]","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:32.902912Z","iopub.execute_input":"2022-11-08T03:37:32.903279Z","iopub.status.idle":"2022-11-08T03:37:36.016698Z","shell.execute_reply.started":"2022-11-08T03:37:32.903248Z","shell.execute_reply":"2022-11-08T03:37:36.014876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame().assign(ENSG00000125257_ABCC4=df6['ENSG00000125257_ABCC4'], CD152=df6['CD152'])","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:36.018631Z","iopub.execute_input":"2022-11-08T03:37:36.018985Z","iopub.status.idle":"2022-11-08T03:37:36.037743Z","shell.execute_reply.started":"2022-11-08T03:37:36.018950Z","shell.execute_reply":"2022-11-08T03:37:36.036660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:36.039854Z","iopub.execute_input":"2022-11-08T03:37:36.040278Z","iopub.status.idle":"2022-11-08T03:37:36.055048Z","shell.execute_reply.started":"2022-11-08T03:37:36.040244Z","shell.execute_reply":"2022-11-08T03:37:36.053580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('out.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:36.056528Z","iopub.execute_input":"2022-11-08T03:37:36.057434Z","iopub.status.idle":"2022-11-08T03:37:36.171762Z","shell.execute_reply.started":"2022-11-08T03:37:36.057374Z","shell.execute_reply":"2022-11-08T03:37:36.170666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/out267/out2.csv')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:36.173265Z","iopub.execute_input":"2022-11-08T03:37:36.173578Z","iopub.status.idle":"2022-11-08T03:37:36.211607Z","shell.execute_reply.started":"2022-11-08T03:37:36.173551Z","shell.execute_reply":"2022-11-08T03:37:36.210711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:37:36.212737Z","iopub.execute_input":"2022-11-08T03:37:36.213880Z","iopub.status.idle":"2022-11-08T03:37:36.225114Z","shell.execute_reply.started":"2022-11-08T03:37:36.213842Z","shell.execute_reply":"2022-11-08T03:37:36.223543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = np.array(df['ENSG00000125257_ABCC4']).reshape(-1, 1)\ny = np.array(df['CD152']).reshape(-1, 1)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:41:52.347287Z","iopub.execute_input":"2022-11-08T03:41:52.347672Z","iopub.status.idle":"2022-11-08T03:41:52.354066Z","shell.execute_reply.started":"2022-11-08T03:41:52.347643Z","shell.execute_reply":"2022-11-08T03:41:52.352362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.25)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:41:55.020642Z","iopub.execute_input":"2022-11-08T03:41:55.020980Z","iopub.status.idle":"2022-11-08T03:41:55.029202Z","shell.execute_reply.started":"2022-11-08T03:41:55.020952Z","shell.execute_reply":"2022-11-08T03:41:55.028284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"regr = LinearRegression()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:44:14.335350Z","iopub.execute_input":"2022-11-08T03:44:14.335744Z","iopub.status.idle":"2022-11-08T03:44:14.341061Z","shell.execute_reply.started":"2022-11-08T03:44:14.335710Z","shell.execute_reply":"2022-11-08T03:44:14.340072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"regr.fit(X_train, y_train)\nprint(regr.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:44:17.167608Z","iopub.execute_input":"2022-11-08T03:44:17.167964Z","iopub.status.idle":"2022-11-08T03:44:17.191796Z","shell.execute_reply.started":"2022-11-08T03:44:17.167939Z","shell.execute_reply":"2022-11-08T03:44:17.190979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = regr.predict(X_test)\nplt.scatter(X_test, y_test, color ='b')\nplt.plot(X_test, y_pred, color ='k')","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:44:22.156015Z","iopub.execute_input":"2022-11-08T03:44:22.156391Z","iopub.status.idle":"2022-11-08T03:44:22.409619Z","shell.execute_reply.started":"2022-11-08T03:44:22.156362Z","shell.execute_reply":"2022-11-08T03:44:22.408716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_binary100 = df[:][:100]","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:47:48.629618Z","iopub.execute_input":"2022-11-08T03:47:48.630111Z","iopub.status.idle":"2022-11-08T03:47:48.637594Z","shell.execute_reply.started":"2022-11-08T03:47:48.630076Z","shell.execute_reply":"2022-11-08T03:47:48.635635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lmplot(x =\"ENSG00000125257_ABCC4\", y =\"CD152\", data = df_binary100,\n                               order = 2, ci = None)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:47:50.646893Z","iopub.execute_input":"2022-11-08T03:47:50.647322Z","iopub.status.idle":"2022-11-08T03:47:50.876624Z","shell.execute_reply.started":"2022-11-08T03:47:50.647289Z","shell.execute_reply":"2022-11-08T03:47:50.875102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_binary100.fillna(method ='ffill', inplace = True)\n  \nX = np.array(df_binary100['ENSG00000125257_ABCC4']).reshape(-1, 1)\ny = np.array(df_binary100['CD152']).reshape(-1, 1)\n  \ndf_binary100.dropna(inplace = True)\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.25)\n  \nregr = LinearRegression()\nregr.fit(X_train, y_train)\nprint(regr.score(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:50:00.892230Z","iopub.execute_input":"2022-11-08T03:50:00.892596Z","iopub.status.idle":"2022-11-08T03:50:00.908486Z","shell.execute_reply.started":"2022-11-08T03:50:00.892571Z","shell.execute_reply":"2022-11-08T03:50:00.907493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = regr.predict(X_test)\nplt.scatter(X_test, y_test, color ='b')\nplt.plot(X_test, y_pred, color ='k')\n  \nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:51:16.169239Z","iopub.execute_input":"2022-11-08T03:51:16.169666Z","iopub.status.idle":"2022-11-08T03:51:16.291156Z","shell.execute_reply.started":"2022-11-08T03:51:16.169629Z","shell.execute_reply":"2022-11-08T03:51:16.290038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error,mean_squared_error\n  \nmae = mean_absolute_error(y_true=y_test,y_pred=y_pred)\n#squared True returns MSE value, False returns RMSE value.\nmse = mean_squared_error(y_true=y_test,y_pred=y_pred) #default=True\nrmse = mean_squared_error(y_true=y_test,y_pred=y_pred,squared=False)\n  \nprint(\"MAE:\",mae)\nprint(\"MSE:\",mse)\nprint(\"RMSE:\",rmse)","metadata":{"execution":{"iopub.status.busy":"2022-11-08T03:51:58.116664Z","iopub.execute_input":"2022-11-08T03:51:58.117033Z","iopub.status.idle":"2022-11-08T03:51:58.127160Z","shell.execute_reply.started":"2022-11-08T03:51:58.116975Z","shell.execute_reply":"2022-11-08T03:51:58.125824Z"},"trusted":true},"execution_count":null,"outputs":[]}]}