{"cells":[{"metadata":{},"cell_type":"markdown","source":"![TurnOff](https://drive.google.com/uc?export=view&id=14iabidS4S0Ur7R5smsHYBLZjTP9M3vWY )\n\nYou cannot use the internet in this competition. Turn it off.\n> このコンペではインターネットを使うことはできません。右下のSettingsからインターネットをOFFにします。","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"%matplotlib inline\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.ticker as ticker\nimport seaborn as sns\nimport glob\nimport numpy as np\nimport pandas as pd\nimport collections","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/landmark-recognition-2020/train.csv\")\ntrain_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"landmark_id = train_df.landmark_id.unique()\nprint(\"Number of landmark_id : \", len(landmark_id))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"imagedata = [] #初期化\nlandmark_id_list = []\n\nfor ID_N in range(80):\n    train_df_id = train_df.loc[train_df.landmark_id == landmark_id[ID_N]]\n    train_df_id = train_df_id.reset_index(drop=True)\n    landmark_id_list.append(landmark_id[ID_N])\n    \n    num1 = str(train_df_id.id[0])[0]\n    num2 = str(train_df_id.id[0])[1]\n    num3 = str(train_df_id.id[0])[2]\n    filename = str(train_df_id.iloc[0, 0])\n    filepath = \"../input/landmark-recognition-2020/train/\" +num1+ \"/\" +num2+\"/\" +num3+ \"/\" + filename + \".jpg\"\n    imagedata.append(cv2.imread(filepath))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(20,24))\n\nfor i in range(80):\n    plt.subplot(8, 10, i+1)\n    img = imagedata[i]\n    plt.title(landmark_id_list[i])\n    plt.grid(False)\n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    plt.tick_params(labelbottom=False,\n                    labelleft=False,\n                    labelright=False,\n                    labeltop=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ID_N = 0\n\ntrain_df_id = train_df.loc[train_df.landmark_id == landmark_id[ID_N]]\ntrain_df_id = train_df_id.reset_index(drop=True)\n\nimagedata = []\n\nfor i in range(len(train_df_id)):\n    num1 = str(train_df_id.id[i])[0]\n    num2 = str(train_df_id.id[i])[1]\n    num3 = str(train_df_id.id[i])[2]\n    filename = str(train_df_id.iloc[i, 0])   \n\n    filepath = \"../input/landmark-recognition-2020/train/\" +num1+ \"/\" +num2+\"/\" +num3+ \"/\" + filename + \".jpg\"\n    imagedata.append(cv2.imread(filepath))\n\nprint(\"landmark_id =\", landmark_id[ID_N])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fig = plt.figure(figsize=(16,8))\n\nfor i in range(12 if len(train_df_id) > 12 else (len(train_df_id))):\n    plt.subplot(3, 4, i+1)\n    img = imagedata[i]\n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"c = collections.Counter(train_df.landmark_id)\nc = collections.Counter(list(c.values()))\nc = sorted(c.items())\n#sns.barplot(c)\n\nx = []\ny = []\n\nfor i in c:\n    x.append(i[0])\n    y.append(i[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.set_palette(\"winter_r\", 8, 0)\n\nfig = plt.figure(figsize=(24, 6))\nax = fig.add_subplot(1, 1, 1)\nsns.barplot(x[:100],y[:100])\nax.set(xlabel ='Number of photos per landmark',ylabel='Number of samples' )\nax.xaxis.set_major_locator(ticker.MultipleLocator(1))\nax.legend()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sns.barplot(x[:19],y[:19])\nplt.xlabel(\"Number of photos per landmark\")\nplt.ylabel(\"Number of samples\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"total1 = 0\ntotal2 = 0\n\nfor i in c:\n    if int(i[0]) < 7:\n        total1 = total1 + int(i[1])\n    else:\n        total2 = total2 + int(i[1])\n\nprint(\"写真が7枚未満の合計:\", total1, \"写真が7枚以上の合計:\", total2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pd.DataFrame(c).describe()[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df = pd.read_csv(\"../input/landmark-recognition-2020/sample_submission.csv\")\nsample_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"sample_df.dtypes","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission = sample_df.copy()\n#submission = submission.drop([\"landmarks\"], axis=1)\n#submission[\"landmarks\"] = \"1000 0.05\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"submission","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}