{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":108394,"databundleVersionId":13172641,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Example dataset(not necessary) /kaggle/input/h690/h690/reference_shapes\n## required images /kaggle/input/h690/h690/sherd_images\n## Dataset /kaggle/input/h690/h690/jd_sherds_info.csv","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\ndf = pd.read_csv('/kaggle/input/h690/h690/jd_sherds_info.csv')\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T10:17:26.304447Z","iopub.execute_input":"2025-08-29T10:17:26.304693Z","iopub.status.idle":"2025-08-29T10:17:26.774115Z","shell.execute_reply.started":"2025-08-29T10:17:26.304675Z","shell.execute_reply":"2025-08-29T10:17:26.773322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head(50)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T10:17:39.291970Z","iopub.execute_input":"2025-08-29T10:17:39.292478Z","iopub.status.idle":"2025-08-29T10:17:39.310411Z","shell.execute_reply.started":"2025-08-29T10:17:39.292454Z","shell.execute_reply":"2025-08-29T10:17:39.309399Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.nunique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T09:31:06.068655Z","iopub.execute_input":"2025-08-29T09:31:06.068980Z","iopub.status.idle":"2025-08-29T09:31:06.111540Z","shell.execute_reply.started":"2025-08-29T09:31:06.068954Z","shell.execute_reply":"2025-08-29T09:31:06.110517Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\nIMG_DIR = '/kaggle/input/h690/h690/sherd_images/'\nfor i, row in df[:5].iterrows():\n    img = cv2.imread(IMG_DIR + row['image_id'] + '.jpg')\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    plt.imshow(img)\n    plt.title(f'type: {row.type}; part: {row.part}; side: {row.image_side} ')\n    plt.axis('off')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T09:47:36.121406Z","iopub.execute_input":"2025-08-29T09:47:36.121810Z","iopub.status.idle":"2025-08-29T09:47:37.739380Z","shell.execute_reply.started":"2025-08-29T09:47:36.121782Z","shell.execute_reply":"2025-08-29T09:47:37.738185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## In fact, I haven't been able to fully figure out how to make a proper submission yet.","metadata":{}},{"cell_type":"markdown","source":"# errors that I have encountered:\n## 1. Submission must have 20 rows\n## 2. ID column image_id not found in submission\n## 3. Submission contains null values\n## 4. Duplicate ID values found in submission\n## 5. MAIN ERROR: Solution and submission values for image_id do not match","metadata":{}},{"cell_type":"markdown","source":"## My submission","metadata":{}},{"cell_type":"code","source":"sherd_groups = {}\nfor _, row in df.iterrows():\n    sherd_id = row['sherd_id']\n    if sherd_id not in sherd_groups:\n        sherd_groups[sherd_id] = {'exterior': [], 'interior': []}\n    \n    if row['image_side'] == 'exterior':\n        sherd_groups[sherd_id]['exterior'].append(row['image_id'])\n    else:\n        sherd_groups[sherd_id]['interior'].append(row['image_id'])\n\nall_sherds = list(sherd_groups.keys())\nnp.random.shuffle(all_sherds)\n\ngroups = []\nused_sherds = set()\n\nfor group_id in range(1, 21):\n    group_size = np.random.randint(1, 2)\n    available_sherds = [s for s in all_sherds if s not in used_sherds]\n    \n    if len(available_sherds) < group_size:\n        available_sherds = all_sherds.copy()\n        used_sherds.clear()\n    \n    group_sherds = available_sherds[:group_size]\n    used_sherds.update(group_sherds)\n\n    exterior_list = []\n    interior_list = []\n    \n    for sherd in group_sherds:\n        exterior_list.extend(sherd_groups[sherd]['exterior'])\n        interior_list.extend(sherd_groups[sherd]['interior'])\n\n    groups.append({\n        'group_id': group_id,\n        'exterior_ids': ';'.join(exterior_list),\n        'interior_ids': ';'.join(interior_list),\n        'image_id': ';'.join(group_sherds)\n    })\n\nsubmission_df = pd.DataFrame(groups)\nsubmission_df.to_csv('MySubmission.csv', index=False)\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T10:03:53.670414Z","iopub.execute_input":"2025-08-29T10:03:53.670735Z","iopub.status.idle":"2025-08-29T10:03:55.404040Z","shell.execute_reply.started":"2025-08-29T10:03:53.670705Z","shell.execute_reply":"2025-08-29T10:03:55.403250Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Success submission","metadata":{}},{"cell_type":"code","source":"submission = []\nall_image_ids = df['image_id'].unique().tolist()\n\nfor i, image_id in enumerate(all_image_ids[:20], 1):\n    sherd_id = df[df['image_id'] == image_id]['sherd_id'].iloc[0]\n    submission.append({\n        'image_id': image_id,\n        'sherd_ids': sherd_id\n    })\n\nsubmission_df = pd.DataFrame(submission)\nsubmission_df.to_csv('submission.csv', index=False)\nsubmission_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-08-29T10:10:56.851789Z","iopub.execute_input":"2025-08-29T10:10:56.852102Z","iopub.status.idle":"2025-08-29T10:10:56.943336Z","shell.execute_reply.started":"2025-08-29T10:10:56.852077Z","shell.execute_reply":"2025-08-29T10:10:56.942391Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}