{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":14420,"databundleVersionId":868327,"sourceType":"competition"}],"dockerImageVersionId":30664,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nprint(os.listdir(\"../input/recursion-cellular-image-classification\"))\nimport sys\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport cv2\nfrom PIL import Image\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2024-03-12T02:45:10.522957Z","iopub.execute_input":"2024-03-12T02:45:10.52334Z","iopub.status.idle":"2024-03-12T02:45:11.078832Z","shell.execute_reply.started":"2024-03-12T02:45:10.523311Z","shell.execute_reply":"2024-03-12T02:45:11.077565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/recursionpharma/rxrx1-utils\nprint ('rxrx1-utils cloned!')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:40:49.554817Z","iopub.execute_input":"2024-03-12T00:40:49.555187Z","iopub.status.idle":"2024-03-12T00:40:51.267304Z","shell.execute_reply.started":"2024-03-12T00:40:49.555159Z","shell.execute_reply":"2024-03-12T00:40:51.265891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:40:58.271741Z","iopub.execute_input":"2024-03-12T00:40:58.272307Z","iopub.status.idle":"2024-03-12T00:40:59.301342Z","shell.execute_reply.started":"2024-03-12T00:40:58.272245Z","shell.execute_reply":"2024-03-12T00:40:59.300224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sys.path.append('rxrx1-utils')\nimport rxrx.io as rio","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:41:02.219951Z","iopub.execute_input":"2024-03-12T00:41:02.220377Z","iopub.status.idle":"2024-03-12T00:41:17.545037Z","shell.execute_reply.started":"2024-03-12T00:41:02.220307Z","shell.execute_reply":"2024-03-12T00:41:17.543878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 基本パスの設定\nbase_path = '/kaggle/input/recursion-cellular-image-classification/train/RPE-05/Plate3'\n\n# 読み込む画像の名前のリストを作成\nimage_names = [f'D19_s2_w{i}.png' for i in range(1, 7)]  # 例: 'D19_s2_w1.png', 'D19_s2_w2.png', ...\n\n# 画像を格納するためのリストを初期化\nimages = []\n\n# 各画像に対してループ処理\nfor image_name in image_names:\n    # 画像の完全なパスを構築\n    image_path = f'{base_path}/{image_name}'\n    \n    # OpenCVを使用して画像を読み込み\n    image = cv2.imread(image_path, cv2.IMREAD_UNCHANGED)\n    \n    # 画像をリストに追加\n    images.append(image)\n\n# 画像をチャンネル方向に積み重ねて一つのテンソルにする\nt = np.stack(images, axis=-1)\n\n# 最終的なテンソルの形状を確認\nprint(f\"テンソルの形状: {t.shape}\")  # 期待される出力: (512, 512, 6)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:41:20.252762Z","iopub.execute_input":"2024-03-12T00:41:20.253545Z","iopub.status.idle":"2024-03-12T00:41:20.378167Z","shell.execute_reply.started":"2024-03-12T00:41:20.253503Z","shell.execute_reply":"2024-03-12T00:41:20.376793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(2, 3, figsize=(24, 16))\n\nfor i, ax in enumerate(axes.flatten()):\n  ax.axis('off')\n  ax.set_title('channel {}'.format(i + 1))\n  _ = ax.imshow(t[:, :, i], cmap='gray')","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:41:28.132865Z","iopub.execute_input":"2024-03-12T00:41:28.133244Z","iopub.status.idle":"2024-03-12T00:41:29.7077Z","shell.execute_reply.started":"2024-03-12T00:41:28.133215Z","shell.execute_reply":"2024-03-12T00:41:29.706502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = rio.convert_tensor_to_rgb(t)\nx.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:41:36.783853Z","iopub.execute_input":"2024-03-12T00:41:36.784507Z","iopub.status.idle":"2024-03-12T00:41:36.891611Z","shell.execute_reply.started":"2024-03-12T00:41:36.784472Z","shell.execute_reply":"2024-03-12T00:41:36.890502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 8))\nplt.axis('off')\n\n_ = plt.imshow(x)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T00:41:41.00644Z","iopub.execute_input":"2024-03-12T00:41:41.006879Z","iopub.status.idle":"2024-03-12T00:41:41.549138Z","shell.execute_reply.started":"2024-03-12T00:41:41.006842Z","shell.execute_reply":"2024-03-12T00:41:41.547897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 基本パスの設定\nbase_path = '/kaggle/input/recursion-cellular-image-classification/train/HUVEC-08/Plate4'\n\n# 読み込む画像の名前のリストを作成\nimage_names = [f'K09_s1_w{i}.png' for i in range(1, 7)]  # 例: 'K09_s1_w1.png', 'K09_s1_w2.png', ...\n\n# 画像を格納するためのリストを初期化\ny = []\n\n# 各画像に対してループ処理\nfor image_name in image_names:\n    # 画像の完全なパスを構築\n    image_path = f'{base_path}/{image_name}'\n    \n    # OpenCVを使用して画像を読み込み\n    image = cv2.imread(image_path, cv2.IMREAD_UNCHANGED)\n    \n    # 画像をリストに追加\n    y.append(image)\n\n# 画像をnumpy配列に変換して平均化\naverage_image = np.mean(y, axis=0).astype(np.uint8)\n    \nplt.figure(figsize=(8, 8))\nplt.axis('off')\n\n_ = plt.imshow(average_image)","metadata":{"execution":{"iopub.status.busy":"2024-03-12T01:33:42.955049Z","iopub.execute_input":"2024-03-12T01:33:42.95549Z","iopub.status.idle":"2024-03-12T01:33:43.398111Z","shell.execute_reply.started":"2024-03-12T01:33:42.955452Z","shell.execute_reply":"2024-03-12T01:33:43.396968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データセットのパスを設定\nbase_path = '/kaggle/input/recursion-cellular-image-classification'\n\n# 各CSVファイルを読み込み、'id_code'をインデックスに設定し、'dataset'カラムを追加\ntrain = pd.read_csv(f'{base_path}/train.csv').set_index('id_code')\ntrain['dataset'] = 'train'\n\ntrain_controls = pd.read_csv(f'{base_path}/train_controls.csv').set_index('id_code')\ntrain_controls['dataset'] = 'train'\n\ntest = pd.read_csv(f'{base_path}/test.csv').set_index('id_code')\ntest['dataset'] = 'test'\n\ntest_controls = pd.read_csv(f'{base_path}/test_controls.csv').set_index('id_code')\ntest_controls['dataset'] = 'test'\n\n# 読み込んだデータセットを縦に結合\nmd = pd.concat([train, train_controls, test, test_controls], axis=0)\n\nmd.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:19:48.859928Z","iopub.execute_input":"2024-03-12T03:19:48.860433Z","iopub.status.idle":"2024-03-12T03:19:48.972842Z","shell.execute_reply.started":"2024-03-12T03:19:48.860389Z","shell.execute_reply":"2024-03-12T03:19:48.971758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"md.index","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:19:53.129937Z","iopub.execute_input":"2024-03-12T03:19:53.130875Z","iopub.status.idle":"2024-03-12T03:19:53.139186Z","shell.execute_reply.started":"2024-03-12T03:19:53.13084Z","shell.execute_reply":"2024-03-12T03:19:53.137995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in md.columns:\n    print (\">> \",i,\"\\t\", md[i].unique())","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:19:55.096002Z","iopub.execute_input":"2024-03-12T03:19:55.096419Z","iopub.status.idle":"2024-03-12T03:19:55.12217Z","shell.execute_reply.started":"2024-03-12T03:19:55.096385Z","shell.execute_reply":"2024-03-12T03:19:55.120816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in ['experiment', 'plate',  'well',  'sirna','dataset', 'well_type']:\n    print (col)\n    print (md[col].value_counts())\n    sns.countplot(y = col,\n              data = md,\n              order = md[col].value_counts().index)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:19:57.457488Z","iopub.execute_input":"2024-03-12T03:19:57.457885Z","iopub.status.idle":"2024-03-12T03:20:09.885798Z","shell.execute_reply.started":"2024-03-12T03:19:57.45785Z","shell.execute_reply":"2024-03-12T03:20:09.884518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_values_count = md.isnull().sum()\nmissing_values_count","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:20:22.102974Z","iopub.execute_input":"2024-03-12T03:20:22.103372Z","iopub.status.idle":"2024-03-12T03:20:22.127288Z","shell.execute_reply.started":"2024-03-12T03:20:22.103338Z","shell.execute_reply":"2024-03-12T03:20:22.126185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"md = md.fillna(0)\nmd['sirna'] = md['sirna'].str.extract('(\\d+)').astype(float)\nmd.head()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:20:44.054033Z","iopub.execute_input":"2024-03-12T03:20:44.054412Z","iopub.status.idle":"2024-03-12T03:20:44.188479Z","shell.execute_reply.started":"2024-03-12T03:20:44.054383Z","shell.execute_reply":"2024-03-12T03:20:44.187256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = md[md['dataset'] == 'train']\ntest_df = md[md['dataset'] == 'test']\n\ntrain_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:20:49.509989Z","iopub.execute_input":"2024-03-12T03:20:49.510395Z","iopub.status.idle":"2024-03-12T03:20:49.539903Z","shell.execute_reply.started":"2024-03-12T03:20:49.510366Z","shell.execute_reply":"2024-03-12T03:20:49.53898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\nplt.title(\"Distribution of SIRNA in the train and test set\")\nsns.distplot(train_df.sirna,color=\"green\", kde=True,bins='auto', label='train')\nsns.distplot(test_df.sirna,color=\"blue\", kde=True, bins='auto', label='test')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:20:53.81538Z","iopub.execute_input":"2024-03-12T03:20:53.815792Z","iopub.status.idle":"2024-03-12T03:20:54.564744Z","shell.execute_reply.started":"2024-03-12T03:20:53.815762Z","shell.execute_reply":"2024-03-12T03:20:54.563408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat1 = 'sirna'\nfig = plt.subplots(figsize=(15, 5))\n\n# train\nplt.subplot(1, 2, 1)\nsns.kdeplot(train_df[feat1][train_df['site'] == 1], shade=False, color=\"b\", label = 'site 1')\nsns.kdeplot(train_df[feat1][train_df['site'] == 2], shade=False, color=\"r\", label = 'site 2')\nplt.title(feat1)\nplt.xlabel('Feature Values')\nplt.ylabel('Probability')\n\n# test\nplt.subplot(1, 2, 2)\nsns.kdeplot(test_df[feat1][test_df['site'] == 1], shade=False, color=\"b\", label = 'site 1')\nsns.kdeplot(test_df[feat1][test_df['site'] == 2], shade=False, color=\"r\", label = 'site 2')\nplt.title(feat1)\nplt.xlabel('Feature Values')\nplt.ylabel('Probability')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-12T03:22:18.101777Z","iopub.execute_input":"2024-03-12T03:22:18.102861Z","iopub.status.idle":"2024-03-12T03:22:18.474587Z","shell.execute_reply.started":"2024-03-12T03:22:18.102825Z","shell.execute_reply":"2024-03-12T03:22:18.473024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}