{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Загружаем библиотеки","metadata":{}},{"cell_type":"code","source":"!pip install -U albumentations","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:22:18.061952Z","iopub.execute_input":"2023-05-20T16:22:18.062352Z","iopub.status.idle":"2023-05-20T16:22:33.515378Z","shell.execute_reply.started":"2023-05-20T16:22:18.062320Z","shell.execute_reply":"2023-05-20T16:22:33.513760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install opencv-contrib-python==4.5.5.62","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:27:04.731432Z","iopub.execute_input":"2023-05-20T16:27:04.731975Z","iopub.status.idle":"2023-05-20T16:27:22.745841Z","shell.execute_reply.started":"2023-05-20T16:27:04.731924Z","shell.execute_reply":"2023-05-20T16:27:22.744457Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install opencv-python","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:28:00.827697Z","iopub.execute_input":"2023-05-20T16:28:00.828200Z","iopub.status.idle":"2023-05-20T16:28:13.897510Z","shell.execute_reply.started":"2023-05-20T16:28:00.828158Z","shell.execute_reply":"2023-05-20T16:28:13.895873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q --upgrade wandb","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:28:25.971133Z","iopub.execute_input":"2023-05-20T16:28:25.972002Z","iopub.status.idle":"2023-05-20T16:28:43.922725Z","shell.execute_reply.started":"2023-05-20T16:28:25.971955Z","shell.execute_reply":"2023-05-20T16:28:43.921566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install timm","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:37:58.326781Z","iopub.execute_input":"2023-05-20T16:37:58.327253Z","iopub.status.idle":"2023-05-20T16:38:11.577837Z","shell.execute_reply.started":"2023-05-20T16:37:58.327214Z","shell.execute_reply":"2023-05-20T16:38:11.576546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport wandb\nimport shutil \nfrom pathlib import Path\nimport json\n\nimport numpy as np \nimport pandas as pd\nimport random\n\nimport cv2\nfrom tqdm import tqdm\nimport copy\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nimport torch.optim as optim\nimport torchvision\nfrom torchvision import datasets, models, transforms\nfrom torch.optim.lr_scheduler import CosineAnnealingWarmRestarts, CosineAnnealingLR, ReduceLROnPlateau, StepLR\n\ntry:\n    from torchinfo import summary\nexcept:\n    print(\"[INFO] Couldn't find torchinfo... installing it.\")\n    !pip install -q torchinfo\n    from torchinfo import summary\n\nimport timm\n\nimport albumentations as A\nimport albumentations.pytorch as AP\n\nfrom albumentations import (\n    HorizontalFlip, IAAPerspective, ShiftScaleRotate, CLAHE, RandomRotate90, Resize, RandomCrop,\n    Transpose, ShiftScaleRotate, Blur, OpticalDistortion, GridDistortion, HueSaturationValue,\n    IAAAdditiveGaussianNoise, GaussNoise, MotionBlur, MedianBlur, RandomBrightnessContrast, IAAPiecewiseAffine,\n    IAASharpen, IAAEmboss, Flip, OneOf, Compose, Rotate, RandomScale, RandomGridShuffle,\n    RandomContrast, RandomGamma, RandomBrightness, CenterCrop, VerticalFlip, ColorJitter,\n    ChannelShuffle, InvertImg, RGBShift, ElasticTransform, Equalize, RandomResizedCrop, ChannelDropout\n)\n\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom PIL import Image\nimport math\nfrom zipfile import ZipFile","metadata":{"execution":{"iopub.status.busy":"2023-05-20T17:26:52.973271Z","iopub.execute_input":"2023-05-20T17:26:52.973671Z","iopub.status.idle":"2023-05-20T17:26:52.994942Z","shell.execute_reply.started":"2023-05-20T17:26:52.973640Z","shell.execute_reply":"2023-05-20T17:26:52.993979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/noaa-right-whale-recognition/train.csv')\ndf.sample(4)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:51:04.396435Z","iopub.execute_input":"2023-05-20T16:51:04.397480Z","iopub.status.idle":"2023-05-20T16:51:04.458569Z","shell.execute_reply.started":"2023-05-20T16:51:04.397441Z","shell.execute_reply":"2023-05-20T16:51:04.457344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:55:29.895355Z","iopub.execute_input":"2023-05-20T16:55:29.895762Z","iopub.status.idle":"2023-05-20T16:55:29.913765Z","shell.execute_reply.started":"2023-05-20T16:55:29.895732Z","shell.execute_reply":"2023-05-20T16:55:29.912327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.nunique()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:55:34.898618Z","iopub.execute_input":"2023-05-20T16:55:34.899013Z","iopub.status.idle":"2023-05-20T16:55:34.910979Z","shell.execute_reply.started":"2023-05-20T16:55:34.898983Z","shell.execute_reply":"2023-05-20T16:55:34.910145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Из 4544 размеченных изображений уникальных особей - 447. Из них мы будем распознавать на аэрофоснимках самых часто встречаемых особей.","metadata":{}},{"cell_type":"code","source":"grdf = df.groupby(['whaleID'])\ngrdf.size().sort_values(axis=0, ascending=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T16:56:52.011426Z","iopub.execute_input":"2023-05-20T16:56:52.011870Z","iopub.status.idle":"2023-05-20T16:56:52.025074Z","shell.execute_reply.started":"2023-05-20T16:56:52.011838Z","shell.execute_reply":"2023-05-20T16:56:52.023991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"whale_counts = grdf.size().values\nplt.hist(whale_counts, bins=100)\nplt.xlabel('Количество изображений')\nplt.ylabel('Количество китов')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T17:06:19.542250Z","iopub.execute_input":"2023-05-20T17:06:19.542692Z","iopub.status.idle":"2023-05-20T17:06:19.946877Z","shell.execute_reply.started":"2023-05-20T17:06:19.542659Z","shell.execute_reply":"2023-05-20T17:06:19.945537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Размеченная выборка несбалансирована, многие киты встречаются на небольшом количестве изображений. Было решено не выбрасывать из набора китов, изображения которых встречаются редко.","metadata":{}},{"cell_type":"markdown","source":"Посмотрим как выглядят снимки","metadata":{}},{"cell_type":"code","source":"path_im = '/kaggle/input/noaa-right-whale-recognition/imgs.zip'\nwith ZipFile(path_im, 'r') as zip:\n    files = zip.namelist()\n    random_file = random.choice(files)\n    with zip.open(random_file) as img_file:\n        img = Image.open(img_file)\n        plt.imshow(img)\n        plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T17:41:20.632206Z","iopub.execute_input":"2023-05-20T17:41:20.632657Z","iopub.status.idle":"2023-05-20T17:41:21.764243Z","shell.execute_reply.started":"2023-05-20T17:41:20.632622Z","shell.execute_reply":"2023-05-20T17:41:21.763090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_im = '/kaggle/input/noaa-right-whale-recognition/imgs.zip'\nwith ZipFile(path_im, 'r') as zip:\n    files = zip.namelist()\n    random_files = random.sample(files, 6)  # выбираем 6 случайных файлов\n    images = []\n    for file in random_files:\n        with zip.open(file) as img_file:\n            img = Image.open(img_file)\n            images.append(np.array(img))  # добавляем каждое изображение в массив\n    fig, axs = plt.subplots(nrows=2, ncols=3, figsize=(12, 6))  # создаем 2 строки и 3 столбца графиков\n    for i, ax in enumerate(axs.flat):\n        ax.imshow(images[i])  # отображаем каждое изображение в соответствующей ячейке\n        ax.axis('off')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T17:43:40.584163Z","iopub.execute_input":"2023-05-20T17:43:40.585099Z","iopub.status.idle":"2023-05-20T17:43:48.393198Z","shell.execute_reply.started":"2023-05-20T17:43:40.585051Z","shell.execute_reply":"2023-05-20T17:43:48.392029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Узнаем размер картинок","metadata":{}},{"cell_type":"code","source":"path_im = '/kaggle/input/noaa-right-whale-recognition/imgs.zip'\nwith ZipFile(path_im, 'r') as zip:\n    files = zip.namelist()\n    random_files = random.sample(files, 10)  # выбираем 10 случайных файлов\n    images = []\n    for file in random_files:\n        with zip.open(file) as img_file:\n            img = Image.open(img_file)\n            images.append(np.array(img))  # добавляем каждое изображение в массив\n    for im in images:\n        height, width, _ = im.shape # меняем порядок размерности идексов массива\n        print('Размер картинки: {}x{}'.format(width, height))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T17:50:42.833806Z","iopub.execute_input":"2023-05-20T17:50:42.834286Z","iopub.status.idle":"2023-05-20T17:50:44.400947Z","shell.execute_reply.started":"2023-05-20T17:50:42.834255Z","shell.execute_reply":"2023-05-20T17:50:44.400079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Теперь поработаем с файлами разметки носовой части и дыхательного отверстия, предоставленной __","metadata":{}},{"cell_type":"code","source":"# нос\nwith open('/kaggle/input/points1/points1.json') as f:\n    points1 = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T18:29:08.049707Z","iopub.execute_input":"2023-05-20T18:29:08.050536Z","iopub.status.idle":"2023-05-20T18:29:08.075459Z","shell.execute_reply.started":"2023-05-20T18:29:08.050497Z","shell.execute_reply":"2023-05-20T18:29:08.074315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# дыхало\nwith open('/kaggle/input/points2/points2.json') as f:\n    points2 = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T18:29:09.858302Z","iopub.execute_input":"2023-05-20T18:29:09.858693Z","iopub.status.idle":"2023-05-20T18:29:09.882374Z","shell.execute_reply.started":"2023-05-20T18:29:09.858666Z","shell.execute_reply":"2023-05-20T18:29:09.881323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_t = [{'filename': d['filename'],\n         'x_t': d['annotations'][0]['x'],\n         'y_t': d['annotations'][0]['y']}\n        for d in points1]\n        \ndf_t = pd.DataFrame.from_records(df_t)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T18:29:11.479086Z","iopub.execute_input":"2023-05-20T18:29:11.479510Z","iopub.status.idle":"2023-05-20T18:29:11.497660Z","shell.execute_reply.started":"2023-05-20T18:29:11.479481Z","shell.execute_reply":"2023-05-20T18:29:11.496153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_bh = [{'filename': d['filename'],\n         'x_bh': d['annotations'][0]['x'],\n         'y_bh': d['annotations'][0]['y']}\n        for d in points2]\n\ndf_bh = pd.DataFrame.from_records(df_bh)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T18:29:13.583951Z","iopub.execute_input":"2023-05-20T18:29:13.584972Z","iopub.status.idle":"2023-05-20T18:29:14.288738Z","shell.execute_reply.started":"2023-05-20T18:29:13.584922Z","shell.execute_reply":"2023-05-20T18:29:14.287485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_annot = pd.merge(df_t, df_bh, on='filename') # объединяем DataFrame по имени файла\n\nprint(df_annot)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T18:30:53.679486Z","iopub.execute_input":"2023-05-20T18:30:53.679934Z","iopub.status.idle":"2023-05-20T18:30:53.713058Z","shell.execute_reply.started":"2023-05-20T18:30:53.679901Z","shell.execute_reply":"2023-05-20T18:30:53.711826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ID = df.rename(columns = {'Image' : 'filename'})\ndf_ID.sort_values(by='filename', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:09:14.222152Z","iopub.execute_input":"2023-05-20T19:09:14.223351Z","iopub.status.idle":"2023-05-20T19:09:14.249048Z","shell.execute_reply.started":"2023-05-20T19:09:14.223308Z","shell.execute_reply":"2023-05-20T19:09:14.247470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df_ID.shape)\nprint(df_annot.shape)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:09:28.316169Z","iopub.execute_input":"2023-05-20T19:09:28.316566Z","iopub.status.idle":"2023-05-20T19:09:28.323011Z","shell.execute_reply.started":"2023-05-20T19:09:28.316537Z","shell.execute_reply":"2023-05-20T19:09:28.321992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_annot","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:09:32.500348Z","iopub.execute_input":"2023-05-20T19:09:32.500738Z","iopub.status.idle":"2023-05-20T19:09:32.518510Z","shell.execute_reply.started":"2023-05-20T19:09:32.500710Z","shell.execute_reply":"2023-05-20T19:09:32.517158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_ID","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:09:39.630526Z","iopub.execute_input":"2023-05-20T19:09:39.630964Z","iopub.status.idle":"2023-05-20T19:09:39.647099Z","shell.execute_reply.started":"2023-05-20T19:09:39.630930Z","shell.execute_reply":"2023-05-20T19:09:39.645570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_an = pd.merge(df_ID, df_annot, on='filename')\ndf_an","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:12:58.750438Z","iopub.execute_input":"2023-05-20T19:12:58.750893Z","iopub.status.idle":"2023-05-20T19:12:58.783047Z","shell.execute_reply.started":"2023-05-20T19:12:58.750861Z","shell.execute_reply":"2023-05-20T19:12:58.781833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Рассчитываем и добавляем в датафрейм расстояние от носа до выдувного отверстия и угол поворота","metadata":{}},{"cell_type":"code","source":"lenght = []\nfor _, row in df_an.iterrows():\n    x_t = row['x_t']\n    y_t = row['y_t']\n    x_bh = row['x_bh']\n    y_bh = row['y_bh']\n    L = np.linalg.norm(np.array([x_t, y_t]) - np.array([x_bh, y_bh]))\n    lenght.append(L)\n\ndf_an['lenght'] = lenght # добавляем новой колонкой в df\n\nangle = [] # угол между x и lenght\n\nfor _, row in df_an.iterrows():\n    x = row['x_t'] - row['x_bh']\n    L = row['lenght']\n    angle_rad = math.acos(x/L)\n    angle_deg = math.degrees(angle_rad)\n    angle.append(angle_deg)\n\ndf_an['angle'] = angle # добавляем новой колонкой в df","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:34:44.852379Z","iopub.execute_input":"2023-05-20T19:34:44.852779Z","iopub.status.idle":"2023-05-20T19:34:45.636341Z","shell.execute_reply.started":"2023-05-20T19:34:44.852751Z","shell.execute_reply":"2023-05-20T19:34:45.635223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Попробуем кропнуть картинку с центром в точке дыхательного отверстия, длинной и шириной немного большей, чем длина до носа","metadata":{}},{"cell_type":"code","source":"new_size = [] # размер обрезки изображения\n\nfor _, row in df_an.iterrows():\n    L = row['lenght']\n    size = 2.6 * L\n    new_size.append(size)\n\ndf_an['new_size'] = new_size # добавляем новой колонкой в df","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:42:43.651269Z","iopub.execute_input":"2023-05-20T19:42:43.651800Z","iopub.status.idle":"2023-05-20T19:42:43.935277Z","shell.execute_reply.started":"2023-05-20T19:42:43.651762Z","shell.execute_reply":"2023-05-20T19:42:43.933661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_an","metadata":{"execution":{"iopub.status.busy":"2023-05-20T19:43:50.955621Z","iopub.execute_input":"2023-05-20T19:43:50.956020Z","iopub.status.idle":"2023-05-20T19:43:50.976670Z","shell.execute_reply.started":"2023-05-20T19:43:50.955990Z","shell.execute_reply":"2023-05-20T19:43:50.975761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Создаем новую папку\nos.makedirs('/kaggle/working/imgs1')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T20:03:03.812726Z","iopub.execute_input":"2023-05-20T20:03:03.813222Z","iopub.status.idle":"2023-05-20T20:03:03.819292Z","shell.execute_reply.started":"2023-05-20T20:03:03.813160Z","shell.execute_reply":"2023-05-20T20:03:03.818060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\nwith zipfile.ZipFile(path_im, 'r') as zip_ref:\n    zip_ref.extractall('/kaggle/working/imgs1')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T20:19:27.508949Z","iopub.execute_input":"2023-05-20T20:19:27.509427Z","iopub.status.idle":"2023-05-20T20:22:05.761096Z","shell.execute_reply.started":"2023-05-20T20:19:27.509394Z","shell.execute_reply":"2023-05-20T20:22:05.758380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = \"/kaggle/working/imgs1/imgs\"\nfiles = os.listdir(folder_path)\nnum_files = len(files)\nprint(\"В папке {} файлов: {}\".format(folder_path, num_files))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T20:51:52.521524Z","iopub.execute_input":"2023-05-20T20:51:52.522745Z","iopub.status.idle":"2023-05-20T20:51:52.538406Z","shell.execute_reply.started":"2023-05-20T20:51:52.522701Z","shell.execute_reply":"2023-05-20T20:51:52.537103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if not os.path.exists('/kaggle/working/marked_imgs'):\n    os.mkdir('/kaggle/working/marked_imgs')\n    \nfolder = '/kaggle/working/imgs1/imgs'\nfiles = os.listdir(folder)\n\n# Фильтруем список файлов по маске и проверяем наличие разметки\nmarked_files = []\nfor file in df_an['filename']:\n    if file.endswith('.jpg') and file in files:\n        marked_files.append(file)\n        \n# Копируем только отмеченные файлы в новую папку\nfor file in marked_files:\n    shutil.copyfile(os.path.join(folder, file), os.path.join('/kaggle/working/marked_imgs', file))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T20:49:59.908500Z","iopub.execute_input":"2023-05-20T20:49:59.909486Z","iopub.status.idle":"2023-05-20T20:50:31.335185Z","shell.execute_reply.started":"2023-05-20T20:49:59.909438Z","shell.execute_reply":"2023-05-20T20:50:31.334111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = \"/kaggle/working/marked_imgs\"\nfiles = os.listdir(folder_path)\nnum_files = len(files)\nprint(\"В папке {} файлов: {}\".format(folder_path, num_files))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T20:52:40.587245Z","iopub.execute_input":"2023-05-20T20:52:40.587787Z","iopub.status.idle":"2023-05-20T20:52:40.599876Z","shell.execute_reply.started":"2023-05-20T20:52:40.587750Z","shell.execute_reply":"2023-05-20T20:52:40.598571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Удаляем исходную папку\n#shutil.rmtree('/kaggle/working/new_imgs')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:29:12.123991Z","iopub.execute_input":"2023-05-20T22:29:12.124493Z","iopub.status.idle":"2023-05-20T22:29:12.367895Z","shell.execute_reply.started":"2023-05-20T22:29:12.124459Z","shell.execute_reply":"2023-05-20T22:29:12.367055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#shutil.rmtree(folder)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Создаем новую папку\nos.makedirs('/kaggle/working/new_imgs')","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:29:30.596343Z","iopub.execute_input":"2023-05-20T22:29:30.596762Z","iopub.status.idle":"2023-05-20T22:29:30.602052Z","shell.execute_reply.started":"2023-05-20T22:29:30.596732Z","shell.execute_reply":"2023-05-20T22:29:30.600988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Обходим каждое изображение и обрезаем его\nfor index, row in df_an.iterrows():\n        \n    # Открываем изображение\n    img_path = '/kaggle/working/marked_imgs/' + row['filename']\n    if not os.path.exists(img_path):\n        continue\n        \n    img = Image.open(img_path)\n    \n    new_size = row['new_size']\n    \n    # Вычисляем левый верхний угол (x1, y1) и правый нижний угол (x2, y2) для обрезки\n    x1 = row['x_bh'] - new_size/2\n    y1 = row['y_bh'] - new_size/2\n    x2 = x1 + new_size\n    y2 = y1 + new_size\n    \n    # Обрезаем изображение\n    cropped_img = img.crop((x1, y1, x2, y2))      \n    # Сохраняем обрезанный файл\n    new_img_path = f'/kaggle/working/new_imgs/{row[\"filename\"]}'\n    cropped_img.save(new_img_path)","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:30:58.335731Z","iopub.execute_input":"2023-05-20T22:30:58.336747Z","iopub.status.idle":"2023-05-20T22:30:58.771273Z","shell.execute_reply.started":"2023-05-20T22:30:58.336707Z","shell.execute_reply":"2023-05-20T22:30:58.769907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder_path = \"/kaggle/working/marked_imgs\"\nfiles = os.listdir(folder_path)\nnum_files = len(files)\nprint(\"В папке {} файлов: {}\".format(folder_path, num_files))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T21:47:39.252867Z","iopub.execute_input":"2023-05-20T21:47:39.254110Z","iopub.status.idle":"2023-05-20T21:47:39.265637Z","shell.execute_reply.started":"2023-05-20T21:47:39.254049Z","shell.execute_reply":"2023-05-20T21:47:39.263832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Приведем в соответствие разметку и файлы в папке","metadata":{}},{"cell_type":"code","source":"# Создадим список имен файлов из папки /kaggle/working/new_imgs\nexisting_files = os.listdir('/kaggle/working/new_imgs')\n\n# Создадим множество номеров файлов в df_an\ndf_filenames = list(df_an['filename'].values)\n\n# Cписок файлов в df_an, которых нет в папке new_imgs\nmissing_files = [file for file in df_filenames if file not in existing_files]\n\n# Cписок файлов в папке new_imgs, которых нет в df_an\nextra_files = [file for file in existing_files if file not in df_filenames]","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:01:23.618217Z","iopub.execute_input":"2023-05-20T22:01:23.618743Z","iopub.status.idle":"2023-05-20T22:01:24.331945Z","shell.execute_reply.started":"2023-05-20T22:01:23.618706Z","shell.execute_reply":"2023-05-20T22:01:24.330530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"missing_files","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:02:07.515630Z","iopub.execute_input":"2023-05-20T22:02:07.516140Z","iopub.status.idle":"2023-05-20T22:02:07.523957Z","shell.execute_reply.started":"2023-05-20T22:02:07.516092Z","shell.execute_reply":"2023-05-20T22:02:07.522608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_an = df_an[~df_an['filename'].isin(missing_files)]\ndf_an.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:02:49.960748Z","iopub.execute_input":"2023-05-20T22:02:49.961209Z","iopub.status.idle":"2023-05-20T22:02:49.975835Z","shell.execute_reply.started":"2023-05-20T22:02:49.961173Z","shell.execute_reply":"2023-05-20T22:02:49.974375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Как выглядят обрезанные картинки","metadata":{}},{"cell_type":"code","source":"path_dir = '/kaggle/working/marked_imgs'\nfiles = os.listdir(path_dir)\nrandom_files = random.sample(files, 6)  # выбираем 6 случайных файлов\n\nimages = []\nfor file in random_files:\n    with Image.open(os.path.join(path_dir, file)) as img:\n        images.append(np.array(img))  # добавляем каждое изображение в массив\n            \nfig, axs = plt.subplots(nrows=2, ncols=3, figsize=(12, 6))  # создаем 2 строки и 3 столбца графиков\nfor i, ax in enumerate(axs.flat):\n    ax.imshow(images[i])  # отображаем каждое изображение в соответствующей ячейке\n    ax.axis('off')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:17:01.639441Z","iopub.execute_input":"2023-05-20T22:17:01.639911Z","iopub.status.idle":"2023-05-20T22:17:10.204636Z","shell.execute_reply.started":"2023-05-20T22:17:01.639876Z","shell.execute_reply":"2023-05-20T22:17:10.203432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path_dir = '/kaggle/working/marked_imgs'\nfiles = os.listdir(path_dir)\nrandom_files = random.sample(files, 10)  # выбираем 10 случайных файлов\n\nimages = []\nfor file in random_files:\n    with Image.open(os.path.join(path_dir, file)) as img:\n        images.append(np.array(img))  # добавляем каждое изображение в массив\n    \nfor im in images:\n    height, width, _ = im.shape # меняем порядок размерности идексов массива\n    print('Размер картинки: {}x{}'.format(width, height))","metadata":{"execution":{"iopub.status.busy":"2023-05-20T22:17:27.997602Z","iopub.execute_input":"2023-05-20T22:17:27.998082Z","iopub.status.idle":"2023-05-20T22:17:28.867398Z","shell.execute_reply.started":"2023-05-20T22:17:27.998047Z","shell.execute_reply":"2023-05-20T22:17:28.866025Z"},"trusted":true},"execution_count":null,"outputs":[]}]}