{"cells":[{"metadata":{"trusted":true},"cell_type":"code","source":"%%capture\n!pip install wandb --upgrade","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\n\nimport matplotlib.pyplot as plt\n\nimport wandb\nfrom kaggle_secrets import UserSecretsClient\n\nuser_secrets = UserSecretsClient()\nwandb_api = user_secrets.get_secret(\"wandb_api\")\n\nwandb.login(key=wandb_api)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Raw HPA Dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"# download RAW dataset csv\nrun = wandb.init(project='hpa', job_type='consume_raw')\nartifact = run.use_artifact('ayush-thakur/hpa/raw:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_df = pd.read_csv(artifact_dir+'/'+'train.csv')\nraw_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Raw Single Label Cell Level Dataset"},{"metadata":{"trusted":true},"cell_type":"code","source":"# download RAW dataset csv\nrun = wandb.init(project='hpa', job_type='consume_single_label_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/single_label_cell_level:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls ./artifacts/single_label_cell_level:v0","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_single_label_df = pd.read_csv(artifact_dir+'/'+'single_label_cell_level.csv')\nraw_single_label_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# We can either use protein or rgb directory.\nsingle_label_cell_level_path = '../input/hpa-single-label-cell-level-dataset/single-label-cell-level/rgb'\nprint(len(os.listdir(single_label_cell_level_path)))\nprint(len(raw_single_label_df))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = os.listdir(single_label_cell_level_path)\nfile_names[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Ref: https://www.kaggle.com/divyanshuusingh/eda-image-segmentation\nlabel_names= {\n0: \"Nucleoplasm\",\n1: \"Nuclear membrane\",\n2: \"Nucleoli\",\n3: \"Nucleoli fibrillar center\",\n4: \"Nuclear speckles\",\n5: \"Nuclear bodies\",\n6: \"Endoplasmic reticulum\",\n7: \"Golgi apparatus\",\n8: \"Intermediate filaments\",\n9: \"Actin filaments\",\n10: \"Microtubules\",\n11: \"Mitotic spindle\",\n12: \"Centrosome\",\n13: \"Plasma membrane\",\n14: \"Mitochondria\",\n15: \"Aggresome\",\n16: \"Cytosol\",\n17: \"Vesicles and punctate cytosolic patterns\",\n18: \"Negative\"\n}\n\nlabels, counts = np.unique(raw_single_label_df.Label.values, return_counts=True)\nprint(f'The unique labels are: {labels} and there values are: {counts}')\n\nplt.figure(figsize=(15,5))\nplt.bar(labels, counts)\n\nfor index, value in enumerate(counts):\n    plt.text(index-0.25, value, str(value), fontdict=dict(fontsize=10))\n\nplt.xticks(np.arange(len(labels)), labels=label_names.values(), rotation=85)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Download Extra Pubic Data "},{"metadata":{},"cell_type":"markdown","source":"### Raw CSV"},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='consume_public_hpa_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/hpa_public_data:v1', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_public_df = pd.read_csv(artifact_dir+'/public_hpa.csv')\nraw_public_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Negative Class"},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='consume_public_hpa_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/negative:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"negative_df = pd.read_csv(artifact_dir+'/pubic_negative.csv')\nnegative_df['Image'] = negative_df['Image'].apply(lambda id: id.split('/')[-1])\nnegative_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = os.listdir('../input/singlelabelpublicnegative/protein/')\n\nnegative_label_df = pd.DataFrame(columns = raw_df.columns)\n\nfor i, filename in tqdm(enumerate(file_names)):\n    img_id = '_'.join(filename.split('.')[0].split('_')[:-1])\n    label = int(negative_df.loc[negative_df.Image == img_id].Label_idx.values[0])\n    negative_label_df.loc[i] = [filename, label]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"negative_label_df.head()\nnegative_label_df.to_csv('clean_public_negative.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='public_negative')\nartifact_ = run.use_artifact('ayush-thakur/hpa/negative:v0', type='dataset')\nartifact = wandb.Artifact('negative', type='dataset')\nartifact.add_file('clean_public_negative.csv')\nrun.log_artifact(artifact)\nrun.join()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Aggresome"},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='consume_public_hpa_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/aggresome:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"aggresome_df = pd.read_csv(artifact_dir+'/pubic_aggresome.csv')\naggresome_df['Image'] = aggresome_df['Image'].apply(lambda id: id.split('/')[-1])\naggresome_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = os.listdir('../input/singlelabelpublicaggresome/rgb')\n\ntmp_df = pd.DataFrame(columns = raw_df.columns)\n\nfor i, filename in tqdm(enumerate(file_names)):\n    img_id = '_'.join(filename.split('.')[0].split('_')[:-1])\n    label = int(aggresome_df.loc[aggresome_df.Image == img_id].Label_idx.values[0])\n    tmp_df.loc[i] = [filename, label]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df.head()\ntmp_df.to_csv('clean_public_aggresome.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='public_aggresome')\nartifact = run.use_artifact('ayush-thakur/hpa/aggresome:v0', type='dataset')\nartifact = wandb.Artifact('aggresome', type='dataset')\nartifact.add_file('clean_public_aggresome.csv')\nrun.log_artifact(artifact)\nrun.join()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Nucleur Membrane"},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='consume_public_hpa_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/nuclear_membrane:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls {artifact_dir}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"public_df = pd.read_csv(artifact_dir+'/pubic_nuclear_membrane.csv')\npublic_df['Image'] = public_df['Image'].apply(lambda id: id.split('/')[-1])\npublic_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = os.listdir('../input/singlelabelpublicnuclearmembrane/rgb')\n\ntmp_df = pd.DataFrame(columns = raw_df.columns)\n\nfor i, filename in tqdm(enumerate(file_names)):\n    img_id = '_'.join(filename.split('.')[0].split('_')[:-1])\n    label = int(public_df.loc[public_df.Image == img_id].Label_idx.values[0])\n    tmp_df.loc[i] = [filename, label]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df.to_csv('clean_pubic_nuclear_membrane.csv')\ntmp_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='public_nuclear_membrane')\nartifact = run.use_artifact('ayush-thakur/hpa/nuclear_membrane:v0', type='dataset')\nartifact = wandb.Artifact('nuclear_membrane', type='dataset')\nartifact.add_file('clean_pubic_nuclear_membrane.csv')\nrun.log_artifact(artifact)\nrun.join()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Mitotic Spindle"},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='consume_public_hpa_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/mitotic_spindle:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls {artifact_dir}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"public_df = pd.read_csv(artifact_dir+'/pubic_mitotic_spindle.csv')\npublic_df['Image'] = public_df['Image'].apply(lambda id: id.split('/')[-1])\npublic_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = os.listdir('../input/singlelabelpublicmitoticspindle/protein')\n\ntmp_df = pd.DataFrame(columns = raw_df.columns)\n\nfor i, filename in tqdm(enumerate(file_names)):\n    img_id = '_'.join(filename.split('.')[0].split('_')[:-1])\n    try:\n        label = int(public_df.loc[public_df.Image == img_id].Label_idx.values[0])\n        tmp_df.loc[i] = [filename, label]\n    except:\n        print(filename)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df.to_csv('clean_pubic_mitotic_spindle.csv')\ntmp_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='public_mitotic_spindle')\nartifact = run.use_artifact('ayush-thakur/hpa/mitotic_spindle:v0', type='dataset')\nartifact = wandb.Artifact('mitotic_spindle', type='dataset')\nartifact.add_file('clean_pubic_mitotic_spindle.csv')\nrun.log_artifact(artifact)\nrun.join()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Actin Filament"},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='consume_public_hpa_dataset')\nartifact = run.use_artifact('ayush-thakur/hpa/actin_filaments:v0', type='dataset')\nartifact_dir = artifact.download()\nrun.finish()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!ls {artifact_dir}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"public_df = pd.read_csv(artifact_dir+'/pubic_actin_filaments.csv')\npublic_df['Image'] = public_df['Image'].apply(lambda id: id.split('/')[-1])\npublic_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"file_names = os.listdir('../input/singlelabelpublicactinfilaments/rgb')\n\ntmp_df = pd.DataFrame(columns = raw_df.columns)\n\nfor i, filename in tqdm(enumerate(file_names)):\n    img_id = '_'.join(filename.split('.')[0].split('_')[:-1])\n    try:\n        label = int(public_df.loc[public_df.Image == img_id].Label_idx.values[0])\n        tmp_df.loc[i] = [filename, label]\n    except:\n        print(filename)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"tmp_df.to_csv('clean_pubic_actin_filaments.csv')\ntmp_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='public_actin_filaments')\nartifact = run.use_artifact('ayush-thakur/hpa/actin_filaments:v0', type='dataset')\nartifact = wandb.Artifact('actin_filaments', type='dataset')\nartifact.add_file('clean_pubic_actin_filaments.csv')\nrun.log_artifact(artifact)\nrun.join()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Merge All The Data (HuHa)"},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"raw_single_label_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"nuclear_membrane_df = pd.read_csv('./clean_pubic_nuclear_membrane.csv')\nmitotic_spindle_df = pd.read_csv('./clean_pubic_mitotic_spindle.csv')\nactin_fragment_df = pd.read_csv('./clean_pubic_actin_filaments.csv')\naggresome_df = pd.read_csv('./clean_public_aggresome.csv')\nnegative_df = pd.read_csv('./clean_public_negative.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"negative_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfs = [raw_single_label_df, \n       nuclear_membrane_df, \n       mitotic_spindle_df, \n       actin_fragment_df,\n       aggresome_df,\n       negative_df]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"single_label_cell_level_df = pd.concat(dfs, ignore_index=True)\nsingle_label_cell_level_df = single_label_cell_level_df[['ID', 'Label']]\nsingle_label_cell_level_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"single_label_cell_level_df = single_label_cell_level_df.sample(frac=1).reset_index(drop=True)\nsingle_label_cell_level_df.head(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Ref: https://www.kaggle.com/divyanshuusingh/eda-image-segmentation\nlabel_names= {\n0: \"Nucleoplasm\",\n1: \"Nuclear membrane\",\n2: \"Nucleoli\",\n3: \"Nucleoli fibrillar center\",\n4: \"Nuclear speckles\",\n5: \"Nuclear bodies\",\n6: \"Endoplasmic reticulum\",\n7: \"Golgi apparatus\",\n8: \"Intermediate filaments\",\n9: \"Actin filaments\",\n10: \"Microtubules\",\n11: \"Mitotic spindle\",\n12: \"Centrosome\",\n13: \"Plasma membrane\",\n14: \"Mitochondria\",\n15: \"Aggresome\",\n16: \"Cytosol\",\n17: \"Vesicles and punctate cytosolic patterns\",\n18: \"Negative\"\n}\n\nlabels, counts = np.unique(single_label_cell_level_df.Label.values, return_counts=True)\nprint(f'The unique labels are: {labels} and there values are: {counts}')\n\nplt.figure(figsize=(15,5))\nplt.bar(labels, counts)\n\nfor index, value in enumerate(counts):\n    plt.text(index-0.25, value, str(value), fontdict=dict(fontsize=10))\n\nplt.xticks(np.arange(len(labels)), labels=label_names.values(), rotation=85)\nplt.show()\n\nlabels, counts = np.unique(raw_single_label_df.Label.values, return_counts=True)\nprint(f'The unique labels are: {labels} and there values are: {counts}')\n\nplt.figure(figsize=(15,5))\nplt.bar(labels, counts)\n\nfor index, value in enumerate(counts):\n    plt.text(index-0.25, value, str(value), fontdict=dict(fontsize=10))\n\nplt.xticks(np.arange(len(labels)), labels=label_names.values(), rotation=85)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"single_label_cell_level_df.to_csv('single_label_cell_level_full_dataset.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"run = wandb.init(project='hpa', job_type='slcl_full_dataset_creation')\n\n_ = run.use_artifact('ayush-thakur/hpa/single_label_cell_level:v0', type='dataset')\n_ = run.use_artifact('ayush-thakur/hpa/negative:v1', type='dataset')\n_ = run.use_artifact('ayush-thakur/hpa/aggresome:v1', type='dataset')\n_ = run.use_artifact('ayush-thakur/hpa/actin_filaments:v1', type='dataset')\n_ = run.use_artifact('ayush-thakur/hpa/mitotic_spindle:v1', type='dataset')\n_ = run.use_artifact('ayush-thakur/hpa/nuclear_membrane:v1', type='dataset')\n\nartifact = wandb.Artifact('slcl_full_dataset', type='dataset')\nartifact.add_file('single_label_cell_level_full_dataset.csv')\nrun.log_artifact(artifact)\nrun.join()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}