{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn import feature_extraction, linear_model, model_selection, preprocessing\nimport plotly.graph_objs as go\nimport plotly.offline as py\nimport plotly.express as px\n\n#Ignore warnings\nimport warnings\nwarnings.filterwarnings('ignore')\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-06T22:45:28.715382Z","iopub.execute_input":"2022-07-06T22:45:28.715855Z","iopub.status.idle":"2022-07-06T22:45:28.727897Z","shell.execute_reply.started":"2022-07-06T22:45:28.715818Z","shell.execute_reply":"2022-07-06T22:45:28.726173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<center style=\"font-family:verdana;\"><h1 style=\"font-size:200%; padding: 10px; background: black;\"><b style=\"color:white;\">BioKit Viz</b></h1></center>","metadata":{}},{"cell_type":"markdown","source":"\"BioKit is a set of tools dedicated to bioinformatics, data visualisation (biokit.viz), access to online biological data (e.g. UniProt, NCBI thanks to bioservices). It also contains more advanced tools related to data analysis (e.g., biokit.stats). Since R is quite common in bioinformatics, we also provide a convenient module to run R inside your Python scripts or shell (:mod:biokit.rtools module).\"\n\nhttps://biokit.readthedocs.io/en/latest/","metadata":{}},{"cell_type":"markdown","source":"![](https://pbs.twimg.com/media/FTGWJMGXsAY3ecY.jpg)mobile.twitter.com","metadata":{}},{"cell_type":"code","source":"# Get the unified BCG Strain spreadsheet\ndata = pd.read_csv('/kaggle/input/tabular-playground-series-jul-2022/data.csv', delimiter=',')\npd.set_option('display.max_columns', None)\ndata.head()","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:44:36.869478Z","iopub.execute_input":"2022-07-06T22:44:36.869901Z","iopub.status.idle":"2022-07-06T22:44:38.521077Z","shell.execute_reply.started":"2022-07-06T22:44:36.869870Z","shell.execute_reply":"2022-07-06T22:44:38.519772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install biokit","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-06T22:44:44.330349Z","iopub.execute_input":"2022-07-06T22:44:44.330770Z","iopub.status.idle":"2022-07-06T22:45:13.151993Z","shell.execute_reply.started":"2022-07-06T22:44:44.330738Z","shell.execute_reply":"2022-07-06T22:45:13.150124Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Biokit Generic statistical tools","metadata":{}},{"cell_type":"markdown","source":"#Automatic Adaptative Mixture Fitting","metadata":{}},{"cell_type":"code","source":"#Code by https://biokit.readthedocs.io/en/latest/references.html#biokit.viz.hist2d.Hist2D\n\nfrom biokit.stats.mixture import AdaptativeMixtureFitting, GaussianMixture\nm = GaussianMixture(mu=[-1,1], sigma=[0.5,0.5], mixture=[0.2,0.8])\namf = AdaptativeMixtureFitting(m.data)\namf.run(kmin=1, kmax=6)\namf.diagnostic(k=amf.best_k);","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:59:41.005111Z","iopub.execute_input":"2022-07-06T22:59:41.005582Z","iopub.status.idle":"2022-07-06T22:59:55.167099Z","shell.execute_reply.started":"2022-07-06T22:59:41.005549Z","shell.execute_reply":"2022-07-06T22:59:55.165653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Expectation minimization class to estimate parameters of GMM","metadata":{}},{"cell_type":"code","source":"#Code by https://biokit.readthedocs.io/en/latest/references.html#biokit.viz.hist2d.Hist2D\n\nfrom biokit.stats.mixture import GaussianMixture, EM\nm = GaussianMixture(mu=[-1,1], sigma=[0.5,0.5], mixture=[0.2,0.8])\nem = EM(m.data)\nem.estimate(k=2)\nem.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T23:00:42.299106Z","iopub.execute_input":"2022-07-06T23:00:42.299648Z","iopub.status.idle":"2022-07-06T23:00:43.114313Z","shell.execute_reply.started":"2022-07-06T23:00:42.299608Z","shell.execute_reply":"2022-07-06T23:00:43.112617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Creates a mix of Gaussian distribution","metadata":{}},{"cell_type":"code","source":"#Code by https://biokit.readthedocs.io/en/latest/references.html#biokit.viz.hist2d.Hist2D\n\nfrom biokit.stats.mixture import GaussianMixture\nm = GaussianMixture(mu=[-1,1], sigma=[0.5, 0.5], mixture=[.2, .8], N=1000)\nm.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T23:01:17.197135Z","iopub.execute_input":"2022-07-06T23:01:17.197610Z","iopub.status.idle":"2022-07-06T23:01:17.438688Z","shell.execute_reply.started":"2022-07-06T23:01:17.197575Z","shell.execute_reply":"2022-07-06T23:01:17.437562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"GaussianMixtureFitting using scipy minization","metadata":{}},{"cell_type":"code","source":"#Code by https://biokit.readthedocs.io/en/latest/references.html#biokit.viz.hist2d.Hist2D\n\nfrom biokit.stats.mixture import GaussianMixture, GaussianMixtureFitting\nm = GaussianMixture(mu=[-1,1], sigma=[0.5,0.5], mixture=[0.2,0.8])\nmf = GaussianMixtureFitting(m.data)\nmf.estimate(k=2)\nmf.plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-06T23:01:54.517614Z","iopub.execute_input":"2022-07-06T23:01:54.517995Z","iopub.status.idle":"2022-07-06T23:01:55.420572Z","shell.execute_reply.started":"2022-07-06T23:01:54.517964Z","shell.execute_reply":"2022-07-06T23:01:55.419739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#I applied Shi L. snippet to reduce the number of features. So that, Biokit details appears","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport matplotlib.colors\n\n#Code by SHI LONG ZHUANG  https://www.kaggle.com/code/shilongzhuang/intro-to-mice-an-imputation-strategy\n\n\nf4 = data[['f_00', 'f_01', 'f_02', 'f_03', 'f_04', 'f_05', 'f_06', 'f_07', 'f_08', 'f_09', 'f_10', 'f_11', 'f_12', 'f_13', 'f_14']]\n\n\nplt.subplots(figsize = (12, 12))\ncmap = matplotlib.colors.LinearSegmentedColormap.from_list(\"\",\n                                                           ['#363062',\n                                                            '#E9D5CA',\n                                                            '#363062',\n                                                           ])\n\nmask = np.triu(np.ones_like(f4.corr() ))\nsns.heatmap(f4.corr(),\n            mask = mask,\n            cmap = cmap,\n            cbar = False,\n            square = True,\n            annot = True,\n            linewidths = 3,\n           );","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:48:08.098534Z","iopub.execute_input":"2022-07-06T22:48:08.098993Z","iopub.status.idle":"2022-07-06T22:48:09.032299Z","shell.execute_reply.started":"2022-07-06T22:48:08.098956Z","shell.execute_reply":"2022-07-06T22:48:09.030611Z"},"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by Mahendra Gundeti https://www.kaggle.com/code/mahendragundeti/abc-analysis/notebook\n\n%pylab inline\nimport pandas as pd\nmatplotlib.rcParams['figure.dpi'] = 145\nmatplotlib.rcParams['figure.figsize'] = (8,6)\nfrom biokit.viz import corrplot\nc = corrplot.Corrplot(f4)\nc.plot(colorbar=False, method='square', shrink=.9 ,rotation=90, upper='circle',grid='grey',\n       fontsize=6,label_color='purple',\n       cmap='RdYlGn')\nplt.show();\n\nmatplotlib.rcParams['figure.dpi'] = 85\n\n# Calculate pairwise-correlation\nmatrix = f4.corr().round(2)\n\n# Create a mask\nmask = np.tril(np.ones_like(matrix, dtype=bool))\n\n# Create a custom divergin palette\ncmap = sns.diverging_palette(250, 15, s=75, l=40,\n                            n=9, center=\"light\", as_cmap=True)\n\nplt.figure(figsize=(16, 14))\nsns.heatmap(matrix, mask=mask, center=0, annot=True,\n            fmt='.2f', square=True, cmap='RdYlGn')\n\nplt.show();","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:48:28.022430Z","iopub.execute_input":"2022-07-06T22:48:28.022973Z","iopub.status.idle":"2022-07-06T22:48:31.681036Z","shell.execute_reply.started":"2022-07-06T22:48:28.022931Z","shell.execute_reply":"2022-07-06T22:48:31.679380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Biokit Method Square","metadata":{}},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nc.plot(colorbar=False, method='square', shrink=.9 ,rotation=45)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:49:06.084424Z","iopub.execute_input":"2022-07-06T22:49:06.084940Z","iopub.status.idle":"2022-07-06T22:49:06.626840Z","shell.execute_reply.started":"2022-07-06T22:49:06.084903Z","shell.execute_reply":"2022-07-06T22:49:06.625736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Biokit Method Text","metadata":{}},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nc.plot(method='text', fontsize=10, colorbar=False)\n# only red to blue colormap is implemented so far","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:49:18.026039Z","iopub.execute_input":"2022-07-06T22:49:18.027383Z","iopub.status.idle":"2022-07-06T22:49:19.449974Z","shell.execute_reply.started":"2022-07-06T22:49:18.027327Z","shell.execute_reply":"2022-07-06T22:49:19.448638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Biokit Method Color","metadata":{}},{"cell_type":"code","source":"c.plot(method='color'); # shrink not available","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:49:47.298192Z","iopub.execute_input":"2022-07-06T22:49:47.299300Z","iopub.status.idle":"2022-07-06T22:49:47.784520Z","shell.execute_reply.started":"2022-07-06T22:49:47.299231Z","shell.execute_reply":"2022-07-06T22:49:47.781504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Biokit Method Pie","metadata":{}},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nc.plot(method='pie', shrink=.9, grid=False)\nfigsize=(16,14);","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:49:59.873050Z","iopub.execute_input":"2022-07-06T22:49:59.873494Z","iopub.status.idle":"2022-07-06T22:50:00.624296Z","shell.execute_reply.started":"2022-07-06T22:49:59.873458Z","shell.execute_reply":"2022-07-06T22:50:00.623375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nc.order(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-06T22:57:22.840099Z","iopub.execute_input":"2022-07-06T22:57:22.840647Z","iopub.status.idle":"2022-07-06T22:57:22.852171Z","shell.execute_reply.started":"2022-07-06T22:57:22.840606Z","shell.execute_reply":"2022-07-06T22:57:22.850640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Biokit Method Circle","metadata":{}},{"cell_type":"code","source":"c.plot(method='circle');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:57:30.890440Z","iopub.execute_input":"2022-07-06T22:57:30.890866Z","iopub.status.idle":"2022-07-06T22:57:31.517401Z","shell.execute_reply.started":"2022-07-06T22:57:30.890834Z","shell.execute_reply":"2022-07-06T22:57:31.515735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#With ellipses, rotation of the patches helps to see the different patterns\n\nI tried to change the colors though I got : incorrect cmap. Then I kept orange/white/green\n\n#I need more colors BioKit!","metadata":{}},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nc.plot(cmap=('orange', 'white', 'green'));","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:57:43.242892Z","iopub.execute_input":"2022-07-06T22:57:43.243351Z","iopub.status.idle":"2022-07-06T22:57:43.873087Z","shell.execute_reply.started":"2022-07-06T22:57:43.243316Z","shell.execute_reply":"2022-07-06T22:57:43.871853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#More tuning with shrink and fontsizes","metadata":{}},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\n#df = [[1,.5,-.5],[.5,1,-.1],[-.5,-.1,1]]\n#c = corrplot.Corrplot(df)\nc.plot(method='square', cmap=('magenta', 'green', 'cyan'), \n       fontsize=10, shrink=1, colorbar=False)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:57:57.188612Z","iopub.execute_input":"2022-07-06T22:57:57.189019Z","iopub.status.idle":"2022-07-06T22:57:57.693994Z","shell.execute_reply.started":"2022-07-06T22:57:57.188987Z","shell.execute_reply":"2022-07-06T22:57:57.692739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Subplots","metadata":{}},{"cell_type":"code","source":"#Code by https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nfig = plt.figure()\n#plt.figure(figsize=(10, 8))\nfig.subplots_adjust(left=0.2, wspace=0.6)\nax1 = fig.add_subplot(221)\nax2 = fig.add_subplot(222)\nax3 = fig.add_subplot(223)\nax4 = fig.add_subplot(224)\n\nimport numpy as np\nc = corrplot.Corrplot(f4)\n#c = corrplot.Corrplot(np.random.rand(10,10))\nc.plot(fig=fig, ax=ax3, colorbar=True, cmap='copper', method='rectangle')\nc.plot(fig=fig, ax=ax4, colorbar=True, cmap='jet', method='color')\nc.plot(fig=fig, ax=ax2, colorbar=True, cmap='hot', method='ellipse')\nc.plot(fig=fig, ax=ax1, colorbar=True, method='pie');","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-07-06T22:58:08.059529Z","iopub.execute_input":"2022-07-06T22:58:08.060057Z","iopub.status.idle":"2022-07-06T22:58:10.729593Z","shell.execute_reply.started":"2022-07-06T22:58:08.060016Z","shell.execute_reply":"2022-07-06T22:58:10.728391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Acknowledgements:\n\nJupyter NBViewer https://nbviewer.org/github/biokit/biokit/blob/master/notebooks/viz/corrplot.ipynb\n\nCorrplot demonstration: biokit https://github.com/biokit/biokit https://pypi.python.org/pypi/biokit\n\nMahendra Gundeti https://www.kaggle.com/code/mahendragundeti/abc-analysis/notebook\n\nShi Long Zhuang https://www.kaggle.com/code/shilongzhuang/intro-to-mice-an-imputation-strategy","metadata":{}},{"cell_type":"markdown","source":"#Biokit Overview\n\nSince it's a Playground (TPS July 2022), next steps with BioKit:\n\n\nbiokit.network Utilities related to networks (e.g., protein)\n\nbiokit.viz Plotting tools\n\nbiokit.rtool utilities related to R language (e.g., RPackageManager, RSession)\n\nbiokit.sequence Sequence related (Generic, DNA, RNA)\n\nbiokit.stats Generic statistical tools\n\nhttps://biokit.readthedocs.io/en/latest/","metadata":{}}]}