{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":91249,"databundleVersionId":11294684,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Create Annotation Text for BFM YOLO\n\n","metadata":{}},{"cell_type":"markdown","source":"    row_id - index of the row\n    tomo_id - unique identifier of the tomogram. Some tomograms in the train set have multiple motors.\n    Motor axis 0 - the z-coordinate of the motor, i.e., which slice it is located on\n    Motor axis 1 - the y-coordinate of the motor\n    Motor axis 2 - the x-coordinate of the motor\n    Array shape axis 0 - z-axis length, i.e., number of slices in the tomogram\n    Array shape axis 1 - y-axis length, or width of each slice\n    Array shape axis 2 - x-axis length, or height of each slice\n    Voxel spacing - scaling of the tomogram; angstroms per voxel\n    Number of motors - Number of motors in the tomogram. Note that each row represents a motor, so tomograms with multiple motors will have several rows to locate each motor.","metadata":{}},{"cell_type":"code","source":"from PIL import Image\nimport glob\nimport pandas as pd\nimport numpy as np\nimport random\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\n!mkdir train\nimport shutil\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T02:51:38.708712Z","iopub.execute_input":"2025-03-15T02:51:38.709196Z","iopub.status.idle":"2025-03-15T02:51:40.658777Z","shell.execute_reply.started":"2025-03-15T02:51:38.709141Z","shell.execute_reply":"2025-03-15T02:51:40.656984Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df=pd.read_csv('/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/train_labels.csv')\nprint(df.columns.tolist())\ndf=df[df['Number of motors']!=0]\ndisplay(df)\nunique_names = df['tomo_id'].unique().tolist()\nprint(unique_names[0:3])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T02:51:40.660237Z","iopub.execute_input":"2025-03-15T02:51:40.661067Z","iopub.status.idle":"2025-03-15T02:51:40.731167Z","shell.execute_reply.started":"2025-03-15T02:51:40.661016Z","shell.execute_reply":"2025-03-15T02:51:40.730010Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = df.rename(columns={\"Array shape (axis 1)\": \"W\", \"Array shape (axis 2)\": \"H\",\n                        \"Motor axis 1\":\"Yc\", \"Motor axis 2\":\"Xc\"})\ndf['xc']=df['Xc']/df['H']\ndf['yc']=df['Yc']/df['W']\ndf['width']=10/df['W']\ndf['height']=10/df['H']\n#yolo txt shows class_id x_center y_center width height in order.\n\ndf['txt0']=df[['xc','yc','width','height']].apply(lambda row: \" \".join(map(str,row)), axis=1)\ndf['txt']=['0 ']+df['txt0'].astype(str)\ndisplay(df[0:2].T)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T02:51:40.732494Z","iopub.execute_input":"2025-03-15T02:51:40.732949Z","iopub.status.idle":"2025-03-15T02:51:40.766441Z","shell.execute_reply.started":"2025-03-15T02:51:40.732912Z","shell.execute_reply":"2025-03-15T02:51:40.765182Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.loc[1,\"txt\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T02:51:40.769173Z","iopub.execute_input":"2025-03-15T02:51:40.769542Z","iopub.status.idle":"2025-03-15T02:51:40.776757Z","shell.execute_reply.started":"2025-03-15T02:51:40.769513Z","shell.execute_reply":"2025-03-15T02:51:40.775673Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"    dir0='/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/train'\n    for name in unique_names:\n        dfi=df[df['tomo_id']==name]\n        z_axs=dfi['Motor axis 0'].tolist()\n        for zi in z_axs:\n            z=int(zi)\n            dfii=dfi[dfi['Motor axis 0']==z]  \n            dfii.loc[:,'txt'].to_csv(f\"train/{name}_{z}.txt\", index=False, header=False)\n            pre_path=os.path.join(dir0,name,f'slice_{(int(z)):04d}.jpg')\n            post_path=f\"train/{name}_{z}.jpg\"\n            shutil.copy(pre_path,post_path)","metadata":{"execution":{"iopub.status.busy":"2025-03-15T02:51:40.778673Z","iopub.execute_input":"2025-03-15T02:51:40.779155Z","iopub.status.idle":"2025-03-15T02:51:40.814849Z","shell.execute_reply.started":"2025-03-15T02:51:40.779124Z","shell.execute_reply":"2025-03-15T02:51:40.813500Z"}}},{"cell_type":"code","source":"dir0='/kaggle/input/byu-locating-bacterial-flagellar-motors-2025/train'\ntrain_dir = \"dataset/train/\"\nval_dir = \"dataset/valid/\"\nos.makedirs(train_dir, exist_ok=True)\nos.makedirs(val_dir, exist_ok=True)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_files = []\nfor name in unique_names:\n    dfi = df[df[\"tomo_id\"] == name]\n    z_axs = dfi[\"Motor axis 0\"].tolist()\n    for zi in z_axs:\n        z = int(zi)\n        dfii = dfi[dfi[\"Motor axis 0\"] == z]\n        label_txt = dfii[\"txt\"]\n        pre_path = os.path.join(dir0, name, f\"slice_{z:04d}.jpg\")\n        post_name = f\"{name}_{z}\"\n        all_files.append((pre_path, post_name, label_txt))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_files, val_files = train_test_split(all_files, test_size=0.2, random_state=42)\n\nfor pre_path, post_name, label_txt in train_files:\n    if os.path.exists(pre_path):  \n        shutil.copy(pre_path, f\"{train_dir}{post_name}.jpg\")\n        label_txt.to_csv(f\"{train_dir}{post_name}.txt\", index=False, header=False)\n\nfor pre_path, post_name, label_txt in val_files:\n    if os.path.exists(pre_path):  \n        shutil.copy(pre_path, f\"{val_dir}{post_name}.jpg\")\n        label_txt.to_csv(f\"{val_dir}{post_name}.txt\", index=False, header=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-15T02:51:40.815950Z","iopub.execute_input":"2025-03-15T02:51:40.816282Z","iopub.status.idle":"2025-03-15T02:51:40.944572Z","shell.execute_reply.started":"2025-03-15T02:51:40.816254Z","shell.execute_reply":"2025-03-15T02:51:40.942979Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}