{"cells":[{"metadata":{},"cell_type":"markdown","source":"<h1 style=\"text-align: center; font-family: Verdana; font-size: 32px; font-style: normal; font-weight: bold; text-decoration: none; text-transform: none; font-variant: small-caps; letter-spacing: 3px; color: #74d5dd; background-color: #ffffff;\">Human Protein Atlas - Single Cell Classification</h1>\n<h2 style=\"text-align: center; font-family: Verdana; font-size: 16px; font-style: normal; font-weight: bold; text-decoration: underline; text-transform: none; letter-spacing: 2px; color: navy; background-color: #ffffff;\">See How To Submit On JUST The Public Part of the Test Set for Easier LB Probing</h2>\n<h5 style=\"text-align: center; font-family: Verdana; font-size: 12px; font-style: normal; font-weight: bold; text-decoration: None; text-transform: none; letter-spacing: 1px; color: black; background-color: #ffffff;\">CREATED BY: DARIEN SCHETTLER</h5>\n\n---\n\nHeavily inspired by the discussion posts and notebooks created by [**Darek Kłeczek**](https://www.kaggle.com/thedrcat). I was having difficulty getting the technique he discussed to work, so I created this notebook to take it back to first principles and investigate why it wasn't working."},{"metadata":{},"cell_type":"markdown","source":"# 1. Submit Empty Prediction"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport ast\n\npath_to_swap_ss_csv=\"/kaggle/input/hpa-single-cell-image-classification/sample_submission.csv\"\npath_to_orig_ss_csv=\"/kaggle/input/hpasubmission009/sample_submission.csv\"\n\norig_ss_df=pd.read_csv(path_to_orig_ss_csv)\nswap_ss_df=pd.read_csv(path_to_swap_ss_csv)\n\nSUBMIT_METHOD = \"baseline_2\" # One of `empty_1` or `baseline_2`\n\nDEMO_SUBMIT_STYLE=len(swap_ss_df)==559\n# DEMO_SUBMIT_STYLE=False\n\nif DEMO_SUBMIT_STYLE:\n    orig_ss_df = orig_ss_df[:5]\n    swap_ss_df = swap_ss_df[:5]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# ###################################### #\n# # I DON'T KNOW WHY THIS DOESN'T WORK # #\n# ###################################### # \n# nan_replacement = orig_ss_df.PredictionString[0] ## GUESSING THIS IS INVALID SOMEHOW\n#\n# ss_df = swap_ss_df.drop(columns=[\"PredictionString\"]) \\\n#                   .merge(orig_ss_df.drop(columns=[\"ImageWidth\", \"ImageHeight\"]), how=\"left\", on=\"ID\") \\\n#                   .fillna(nan_replacement)\n# \n# ss_df.to_csv(\"/kaggle/working/submission.csv\", index=False)\n# ###################################### # ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def create_pred_col(row):\n    \"\"\" Simple function to return the correct prediction string\n    \n    We will want the original public test dataframe submission when it is \n    available. However, we will use the swapped inn submission dataframe\n    when it is not.\n    \n    Args:\n        row (pd.Series): A row in the dataframe\n    \n    Returns:\n        The prediction string\n    \"\"\"\n    if pd.isnull(row.PredictionString_y):\n        return row.PredictionString_x\n    else:\n        return row.PredictionString_y","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss_df = swap_ss_df.merge(orig_ss_df.drop(columns=[\"ImageWidth\", \"ImageHeight\"]), how=\"left\", on=\"ID\")\nss_df[\"PredictionString\"] = ss_df.apply(create_pred_col, axis=1)\nss_df = ss_df.drop(columns=[\"PredictionString_x\", \"PredictionString_y\"])\n\nif SUBMIT_METHOD==\"empty_1\":\n    ss_df.to_csv(\"/kaggle/working/submission.csv\", index=False)\ndisplay(ss_df)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# 2. Submit Baseline Prediction"},{"metadata":{"trusted":true},"cell_type":"code","source":"PUB_SS_W_META_CSV = \"/kaggle/input/hpa-sample-submission-with-extra-metadata/updated_sample_submission.csv\"\npub_ss_df_w_meta = pd.read_csv(PUB_SS_W_META_CSV)\npub_ss_df_w_meta.mask_rles = pub_ss_df_w_meta.mask_rles.apply(lambda x: ast.literal_eval(x))\npub_ss_df_w_meta.mask_bboxes = pub_ss_df_w_meta.mask_bboxes.apply(lambda x: ast.literal_eval(x))\npub_ss_df_w_meta.mask_sub_rles = pub_ss_df_w_meta.mask_sub_rles.apply(lambda x: ast.literal_eval(x))\n\ndisplay(pub_ss_df_w_meta)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"LABELS_TO_GUESS = [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18]\n# CONFIDENCES = [0.1,]*len(LABELS_TO_GUESS)\nCONFIDENCES = [0.0]*len(LABELS_TO_GUESS)\nCONFIDENCES[7]=1.0\nprint(CONFIDENCES)\n\npub_ss_df_w_meta.PredictionString = pub_ss_df_w_meta.mask_sub_rles.apply(\n    lambda x: \" \".join([\" \".join([f\"{lbl} {conf} {rle}\" \\\n                         for lbl,conf in zip(LABELS_TO_GUESS,CONFIDENCES)]).strip() \\\n               for rle in x]).strip()\n)\n\npub_ss_df_w_meta","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pub_ss_df_w_meta.PredictionString[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ss_df = swap_ss_df.merge(pub_ss_df_w_meta.drop(\n    columns=[\"ImageWidth\", \"ImageHeight\", \"mask_rles\", \"mask_sub_rles\", \"mask_bboxes\"]\n), how=\"left\", on=\"ID\")\n\nss_df[\"PredictionString\"] = ss_df.apply(create_pred_col, axis=1)\nss_df = ss_df.drop(columns=[\"PredictionString_x\", \"PredictionString_y\"])\n\nif SUBMIT_METHOD==\"baseline_2\":\n    ss_df.to_csv(\"/kaggle/working/submission.csv\", index=False)\ndisplay(ss_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}