{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Greetings!  \n\nHere is simple graphical EDA for CITESEQ targets  \nAll 140 targets are analyzed with consideration of 'day' and 'cell type' features from metadata file  \n\nPlots for targets distribution (kernel density estimation) are figured below  \nEvery targets divided by cell type to sublopts  \nEvery plot contains 3 curves - 1 per day presented in train dataset  \nNumber of rows per day are not equal as you can see at curve heights  \nMedian values for every curve added to plot  \nIt allows to see how this target behaves from day to day  \nTargets are clipped to exclude outliers by 1% from every side of distribution  \n\nIf this notebook was useful for you - please upvote","metadata":{}},{"cell_type":"code","source":"# LIBRARIES IMPORT\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport gc\n\nimport numpy as np\nimport pandas as pd\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\nfrom tqdm import tqdm\n\nif not os.path.exists('/opt/conda/lib/python3.7/site-packages/tables'):\n    !pip install --quiet tables\n    \n# READ METADATA\nmeta = pd.read_csv(\"../input/open-problems-multimodal/metadata.csv\", index_col = 'cell_id')\nmeta = meta[meta.technology == 'citeseq']\nmeta.drop('technology', axis = 1, inplace = True)\n\n#READ FEATURES\n# X = pd.read_hdf('../input/open-problems-multimodal/train_cite_inputs.h5')\n\n# READ TARGETS\nY = pd.read_hdf('../input/open-problems-multimodal/train_cite_targets.h5')\n\n# REMOVE 2% OF OULIERS\ncols = list(Y.columns)\nfor i in range(140):\n    col = cols[i]\n    v = Y[col]\n    threshold = 1.0\n    m1 = np.percentile(v, threshold)\n    m2 = np.percentile(v, 100 - threshold)\n    v = np.clip(v, m1, m2)\n    Y[col] = v\n    \n# SHRINK META TO TRAIN SIZE\nmeta = meta[meta.index.isin(Y.index)]\n\n# MERGE TARGETS WITH METADATA\ndf = Y.join(meta)\n\n# CELL TYPE LIST = CTL\nctl = list(meta.cell_type.value_counts().index)","metadata":{"_kg_hide-input":true,"_kg_hide-output":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# BUILDING PLOTS\nfor i in range(140):\n    fig, axs = plt.subplots(ncols =  7, figsize=(30, 4))\n    fig.suptitle(Y.columns[i], fontsize = 20)\n    fig.tight_layout(pad=1.5)\n    for j in range(7):\n        ct = ctl[j]\n        temp_df = Y[[cols[i]]].join(meta)\n        temp_df = temp_df[temp_df.cell_type == ct]\n        m2 = temp_df[temp_df.day == 2][cols[i]].median()\n        m3 = temp_df[temp_df.day == 3][cols[i]].median()\n        m4 = temp_df[temp_df.day == 4][cols[i]].median()\n        \n        g = sns.kdeplot(\n        data=temp_df, x=temp_df[cols[i]], hue=\"day\",\n        palette=\"Set1\", alpha = 1, ax = axs[j])\n        \n        g.set_xlabel(ct, fontsize = 14)\n        g.set_ylabel('', fontsize = 14)\n        \n        g.axvline(x=m2, color='tab:red', ls='solid')\n        g.axvline(x=m3, color='tab:blue', ls='dashed')\n        g.axvline(x=m4, color='tab:green', ls='dotted')","metadata":{"_kg_hide-input":true},"execution_count":null,"outputs":[]}]}