{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:50:59.40286Z","iopub.execute_input":"2023-05-26T06:50:59.403241Z","iopub.status.idle":"2023-05-26T06:50:59.408318Z","shell.execute_reply.started":"2023-05-26T06:50:59.403213Z","shell.execute_reply":"2023-05-26T06:50:59.407254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-05-25T19:31:30.862275Z","iopub.execute_input":"2023-05-25T19:31:30.862789Z","iopub.status.idle":"2023-05-25T19:31:30.889939Z","shell.execute_reply.started":"2023-05-25T19:31:30.862746Z","shell.execute_reply":"2023-05-25T19:31:30.888766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:51:02.888633Z","iopub.execute_input":"2023-05-26T06:51:02.888979Z","iopub.status.idle":"2023-05-26T06:51:02.893118Z","shell.execute_reply.started":"2023-05-26T06:51:02.888954Z","shell.execute_reply":"2023-05-26T06:51:02.89221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import zipfile\n\nwith zipfile.ZipFile('../input/avito-demand-prediction/train_jpg_0.zip', 'r') as zip_file:\n    zip_file.extractall('train_jpg_0')","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:41:12.197993Z","iopub.execute_input":"2023-05-26T06:41:12.19832Z","iopub.status.idle":"2023-05-26T06:43:51.50324Z","shell.execute_reply.started":"2023-05-26T06:41:12.198296Z","shell.execute_reply":"2023-05-26T06:43:51.500687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_0 = pd.read_csv(\"/kaggle/input/df-train-0/df_train_0.csv\")\ndf_train_0.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:43:56.280356Z","iopub.execute_input":"2023-05-26T06:43:56.280805Z","iopub.status.idle":"2023-05-26T06:44:03.229266Z","shell.execute_reply.started":"2023-05-26T06:43:56.280775Z","shell.execute_reply":"2023-05-26T06:44:03.227946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_0.image = '/kaggle/working/train_jpg_0/' + df_train_0.image\ndf_train_0 = df_train_0[[\"image\", \"deal_probability\"]]\ndf_train_0.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:44:07.032682Z","iopub.execute_input":"2023-05-26T06:44:07.033951Z","iopub.status.idle":"2023-05-26T06:44:07.278454Z","shell.execute_reply.started":"2023-05-26T06:44:07.033899Z","shell.execute_reply":"2023-05-26T06:44:07.276936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_result = pd.read_csv(\"/kaggle/input/cat-result/cat_result.csv\")\ncat_result.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:44:11.950978Z","iopub.execute_input":"2023-05-26T06:44:11.951387Z","iopub.status.idle":"2023-05-26T06:44:12.169333Z","shell.execute_reply.started":"2023-05-26T06:44:11.951357Z","shell.execute_reply":"2023-05-26T06:44:12.168631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in cat_result.index:\n    cat_result.prediction[i] = cat_result.prediction[i].replace(']', '')\n    cat_result.prediction[i] = cat_result.prediction[i].replace('[', '')\ncat_result.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:44:14.773727Z","iopub.execute_input":"2023-05-26T06:44:14.774715Z","iopub.status.idle":"2023-05-26T06:44:56.131652Z","shell.execute_reply.started":"2023-05-26T06:44:14.774657Z","shell.execute_reply":"2023-05-26T06:44:56.130722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_result = cat_result.astype({'prediction': float})\ncat_result.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:02.303712Z","iopub.execute_input":"2023-05-26T06:45:02.304138Z","iopub.status.idle":"2023-05-26T06:45:02.345107Z","shell.execute_reply.started":"2023-05-26T06:45:02.304102Z","shell.execute_reply":"2023-05-26T06:45:02.344113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_0['cat_result']= cat_result['prediction']\ndf_train_0.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:05.951394Z","iopub.execute_input":"2023-05-26T06:45:05.952079Z","iopub.status.idle":"2023-05-26T06:45:05.966611Z","shell.execute_reply.started":"2023-05-26T06:45:05.951999Z","shell.execute_reply":"2023-05-26T06:45:05.965177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_0 = df_train_0.set_index('image')\ndf_train_0.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:11.555698Z","iopub.execute_input":"2023-05-26T06:45:11.556063Z","iopub.status.idle":"2023-05-26T06:45:11.574127Z","shell.execute_reply.started":"2023-05-26T06:45:11.556033Z","shell.execute_reply":"2023-05-26T06:45:11.572594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_0.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:25.830691Z","iopub.execute_input":"2023-05-26T06:45:25.831113Z","iopub.status.idle":"2023-05-26T06:45:25.844091Z","shell.execute_reply.started":"2023-05-26T06:45:25.831072Z","shell.execute_reply":"2023-05-26T06:45:25.842813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"content = os.listdir('train_jpg_0')\nlen(content)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:29.066192Z","iopub.execute_input":"2023-05-26T06:45:29.06656Z","iopub.status.idle":"2023-05-26T06:45:29.236568Z","shell.execute_reply.started":"2023-05-26T06:45:29.066533Z","shell.execute_reply":"2023-05-26T06:45:29.235368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.remove(\"/kaggle/working/train_jpg_0/4f029e2a00e892aa2cac27d98b52ef8b13d91471f613c8d3c38e3f29d4da0b0c.jpg\")\nos.remove(\"/kaggle/working/train_jpg_0/8513a91e55670c709069b5f85e12a59095b802877715903abef16b7a6f306e58.jpg\")","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:30.467376Z","iopub.execute_input":"2023-05-26T06:45:30.467738Z","iopub.status.idle":"2023-05-26T06:45:30.473736Z","shell.execute_reply.started":"2023-05-26T06:45:30.467713Z","shell.execute_reply":"2023-05-26T06:45:30.472545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"content = os.listdir('train_jpg_0')\nlen(content)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:32.634428Z","iopub.execute_input":"2023-05-26T06:45:32.634806Z","iopub.status.idle":"2023-05-26T06:45:32.823928Z","shell.execute_reply.started":"2023-05-26T06:45:32.634779Z","shell.execute_reply":"2023-05-26T06:45:32.822746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install \"numpy>=1.16.5,<1.23.0\"","metadata":{"execution":{"iopub.status.busy":"2023-05-26T06:45:34.022529Z","iopub.execute_input":"2023-05-26T06:45:34.022889Z","iopub.status.idle":"2023-05-26T06:45:50.312475Z","shell.execute_reply.started":"2023-05-26T06:45:34.022863Z","shell.execute_reply":"2023-05-26T06:45:50.310987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nimport keras.utils as krs_image","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:10:47.260847Z","iopub.execute_input":"2023-05-26T07:10:47.261866Z","iopub.status.idle":"2023-05-26T07:10:47.270568Z","shell.execute_reply.started":"2023-05-26T07:10:47.261791Z","shell.execute_reply":"2023-05-26T07:10:47.268789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Create the arguments for image preprocessing\n# data_gen_args = dict(\n# #     horizontal_flip=True,\n# #     brightness_range=[0.5, 1.5],\n# #     shear_range=10,\n# #     channel_shift_range=50,\n#     rescale= 1. / 255,\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-25T20:18:43.092012Z","iopub.execute_input":"2023-05-25T20:18:43.092787Z","iopub.status.idle":"2023-05-25T20:18:43.097677Z","shell.execute_reply.started":"2023-05-25T20:18:43.09275Z","shell.execute_reply":"2023-05-25T20:18:43.096556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\n# Create an empty data generator\ndatagen = ImageDataGenerator(rescale = 1./255.)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:10:49.576438Z","iopub.execute_input":"2023-05-26T07:10:49.576865Z","iopub.status.idle":"2023-05-26T07:10:49.583839Z","shell.execute_reply.started":"2023-05-26T07:10:49.576837Z","shell.execute_reply":"2023-05-26T07:10:49.582201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:10:50.764924Z","iopub.execute_input":"2023-05-26T07:10:50.765712Z","iopub.status.idle":"2023-05-26T07:10:50.772953Z","shell.execute_reply.started":"2023-05-26T07:10:50.76564Z","shell.execute_reply":"2023-05-26T07:10:50.770902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set, test_valid_set = train_test_split(df_train_0, train_size = 0.8, random_state = 17)\ntest_set, valid_set = train_test_split(test_valid_set, train_size = 0.5, random_state = 17)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:10:52.923127Z","iopub.execute_input":"2023-05-26T07:10:52.923902Z","iopub.status.idle":"2023-05-26T07:10:53.014015Z","shell.execute_reply.started":"2023-05-26T07:10:52.92386Z","shell.execute_reply":"2023-05-26T07:10:53.012976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size_train = 32\nbatch_size_valid = 32\nbatch_size_test = 2000","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:10:54.440944Z","iopub.execute_input":"2023-05-26T07:10:54.441393Z","iopub.status.idle":"2023-05-26T07:10:54.448204Z","shell.execute_reply.started":"2023-05-26T07:10:54.441368Z","shell.execute_reply":"2023-05-26T07:10:54.446542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from glob import glob\n# images_dir = '/kaggle/working/train_jpg_0/'\n# image_file_list = glob(f'{images_dir}/*.jpg', recursive=True)\n# random.shuffle(image_file_list)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_list_train=train_set.index\nimage_list_test=test_set.index\nimage_list_valid=valid_set.index","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:12:22.416408Z","iopub.execute_input":"2023-05-26T07:12:22.416908Z","iopub.status.idle":"2023-05-26T07:12:22.423278Z","shell.execute_reply.started":"2023-05-26T07:12:22.416872Z","shell.execute_reply":"2023-05-26T07:12:22.422251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_height = 224\nimg_width = 224","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:12:23.599262Z","iopub.execute_input":"2023-05-26T07:12:23.599761Z","iopub.status.idle":"2023-05-26T07:12:23.606009Z","shell.execute_reply.started":"2023-05-26T07:12:23.599733Z","shell.execute_reply":"2023-05-26T07:12:23.604555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def custom_generator(images_list, dataframe, batch_size):\n    i = 0\n    while True:\n        batch = {'images': [], 'csv': [], 'labels': []}\n        for b in range(batch_size):\n            if i == len(images_list):\n                i = 0\n                random.shuffle(images_list)\n            # Read image from list and convert to array\n            image_path = images_list[i]\n            # image_name = os.path.basename(image_path).replace('.jpg', '')\n            image = krs_image.load_img(image_path, target_size=(img_height, img_width))\n            #image = datagen.apply_transform(image, data_gen_args)\n            image = krs_image.img_to_array(image)\n\n            # Read data from csv using the name of current image\n            # csv_row = dataframe.loc[image_name, :]\n            # csv_row = dataframe.loc[image_path, :]\n            # csv_row = dataframe.loc[image_path, 'cat_result']\n            # label = csv_row['deal_probability']\n            label =  dataframe.loc[image_path, 'deal_probability']\n            csv_features = dataframe.loc[image_path, 'cat_result']\n            # csv_features = csv_row.drop(labels='deal_probability')\n            # csv_features = csv_row['cat_result']\n            # label = csv_row.deal_probability[0]\n            # csv_features = csv_row.cat_result[0]\n\n            batch['images'].append(image)\n            batch['csv'].append(csv_features)\n            batch['labels'].append(label)\n\n            i += 1\n\n        batch['images'] = np.array(batch['images'])\n        batch['csv'] = np.array(batch['csv'])\n        # Convert labels to categorical values\n        # batch['labels'] = np.eye(num_classes)[batch['labels']]\n\n        yield [batch['images'], batch['csv']], batch['labels']","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:12:30.296343Z","iopub.execute_input":"2023-05-26T07:12:30.296817Z","iopub.status.idle":"2023-05-26T07:12:30.308261Z","shell.execute_reply.started":"2023-05-26T07:12:30.296779Z","shell.execute_reply":"2023-05-26T07:12:30.30634Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator = custom_generator(image_list_train, train_set, batch_size_train)\nvalid_generator = custom_generator(image_list_valid, valid_set, batch_size_valid)\ntest_generator = custom_generator(image_list_test, test_set, batch_size_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:13:06.169635Z","iopub.execute_input":"2023-05-26T07:13:06.17015Z","iopub.status.idle":"2023-05-26T07:13:06.177177Z","shell.execute_reply.started":"2023-05-26T07:13:06.170115Z","shell.execute_reply":"2023-05-26T07:13:06.175932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# построение модели на основе MobileNet\nfrom keras.applications.mobilenet import MobileNet\nfrom keras.layers import GlobalAveragePooling2D, Dense, Dropout, Flatten\nfrom keras.models import Sequential\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:13:08.159211Z","iopub.execute_input":"2023-05-26T07:13:08.159777Z","iopub.status.idle":"2023-05-26T07:13:08.165884Z","shell.execute_reply.started":"2023-05-26T07:13:08.159748Z","shell.execute_reply":"2023-05-26T07:13:08.16446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow_addons.metrics import RSquare\n\nbase_mobilenet_model = MobileNet(input_shape = (224,224,3), include_top = False)\nfor layer in base_mobilenet_model.layers:\n layer.trainable = False\nmodel_1 = Sequential()\nmodel_1.add(base_mobilenet_model)\nmodel_1.add(GlobalAveragePooling2D())\nmodel_1.add(Dropout(0.5))\nmodel_1.add(Dense(512))\nmodel_1.add(Dropout(0.5))\nmodel_1.add(Dense(1, activation = 'relu'))\nmodel_1.compile(optimizer = 'adam', loss = 'mse',metrics = [RSquare()])\nmodel_1.summary()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:19:32.02101Z","iopub.execute_input":"2023-05-26T07:19:32.021474Z","iopub.status.idle":"2023-05-26T07:19:33.029062Z","shell.execute_reply.started":"2023-05-26T07:19:32.021442Z","shell.execute_reply":"2023-05-26T07:19:33.027516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# графическое представление модели\nfrom keras.utils.vis_utils import plot_model\nplot_model(model_1, show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:19:40.333022Z","iopub.execute_input":"2023-05-26T07:19:40.333374Z","iopub.status.idle":"2023-05-26T07:19:40.4135Z","shell.execute_reply.started":"2023-05-26T07:19:40.333344Z","shell.execute_reply":"2023-05-26T07:19:40.411969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_generator_steps = train_set.shape[0] // batch_size_train\nvalid_generator_steps = valid_set.shape[0] // batch_size_valid","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:19:43.468411Z","iopub.execute_input":"2023-05-26T07:19:43.468798Z","iopub.status.idle":"2023-05-26T07:19:43.474139Z","shell.execute_reply.started":"2023-05-26T07:19:43.468769Z","shell.execute_reply":"2023-05-26T07:19:43.472831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_1.fit_generator(generator = train_generator,\n                      steps_per_epoch = train_generator_steps,\n                      epochs = 50,\n                      validation_data = valid_generator,\n                      validation_steps = valid_generator_steps)","metadata":{"execution":{"iopub.status.busy":"2023-05-26T07:19:45.186941Z","iopub.execute_input":"2023-05-26T07:19:45.187344Z","iopub.status.idle":"2023-05-26T07:19:45.511685Z","shell.execute_reply.started":"2023-05-26T07:19:45.187314Z","shell.execute_reply":"2023-05-26T07:19:45.509736Z"},"trusted":true},"execution_count":null,"outputs":[]}]}