{"cells":[{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"import os\nimport pandas as pd","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"base = \"/kaggle/input/landmark-recognition-2020\"\nos.listdir(base)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(os.path.join(base, \"train.csv\"))\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"top_50 = train_df.landmark_id.value_counts()[:50].index.to_list()\ntop_100 = train_df.landmark_id.value_counts()[:100].index.to_list()\nbottom_50 = train_df.landmark_id.value_counts()[-50:].index.to_list()\nbottom_100 = train_df.landmark_id.value_counts()[-100:].index.to_list()\nextreme_50 = train_df.landmark_id.value_counts()[:25].index.to_list() + train_df.landmark_id.value_counts()[-25:].index.to_list()\nextreme_100 = train_df.landmark_id.value_counts()[:50].index.to_list() + train_df.landmark_id.value_counts()[-50:].index.to_list()\nrandom_50 = train_df.landmark_id.sample(n=50)\nrandom_100 = train_df.landmark_id.sample(n=100)\nrandom_500 = train_df.landmark_id.sample(n=500)\n\ndct = {\"top_50\": top_50, \"top_100\": top_100, \"bottom_50\": bottom_50, \"bottom_100\": bottom_100, \n       \"extreme_50\": extreme_50, \"extreme_100\": extreme_100, \"random_50\": random_50, \"random_100\": random_100,\n       \"random_500\": random_500}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from tqdm.notebook import tqdm\nimport cv2\nfrom shutil import copy2\ndef store_images(name, ids):\n    temp = os.path.join(\"/kaggle/working\", name)\n    os.makedirs(temp, exist_ok=True)\n    for lm_id in tqdm(ids):\n        temp_two = os.path.join(temp, str(lm_id))\n        os.makedirs(temp_two, exist_ok=True)\n        for image_id in train_df[train_df.landmark_id == lm_id].id.to_list():\n            src = os.path.join(base, 'train', image_id[0], image_id[1], image_id[2], image_id + '.jpg')\n            dst = os.path.join(temp_two, image_id + '.jpg')\n            copy2(src, dst)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"'''\n会空间不够\nfor key, value in tqdm(dct.items()):\n    store_images(key, value)\n    \n要用哪个的时候调用哪个函数，然后在你的model notebook引用output，这里举得例子是 top_100\n'''\n\nstore_images('random_50', dct['random_50'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}