{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport random\n\nimport numpy as np \nimport pandas as pd \nimport lightgbm as lgb\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential,Model\nfrom tensorflow.keras.layers import Dense,Conv2D,Flatten,Dropout, Input, Concatenate, BatchNormalization\nfrom tensorflow.keras.callbacks import EarlyStopping,ReduceLROnPlateau\nfrom tensorflow.keras.optimizers.schedules import ExponentialDecay\nfrom PIL import Image\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\nfrom glob import glob\n\nimport seaborn as sns\nfrom pathlib import Path","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-10T22:31:53.723773Z","iopub.execute_input":"2022-01-10T22:31:53.724468Z","iopub.status.idle":"2022-01-10T22:31:53.731442Z","shell.execute_reply.started":"2022-01-10T22:31:53.72443Z","shell.execute_reply":"2022-01-10T22:31:53.730456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#source path (where the Pawpularity contest data resides)\npath = '../input/petfinder-pawpularity-score-clean/'\npath_test = '../input/petfinder-pawpularity-score/'\n#Get the metadata (the .csv data) and put it into DataFrames\ntrain_df = pd.read_csv(path + 'train.csv')\ntest_df = pd.read_csv(path + 'test.csv')\n\n#Get the image data (the .jpg data) and put it into lists of filenames\ntrain_jpg = glob(path + \"train/*.jpg\")\ntest_jpg = glob(path + \"test/*.jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:53.733235Z","iopub.execute_input":"2022-01-10T22:31:53.734038Z","iopub.status.idle":"2022-01-10T22:31:54.016867Z","shell.execute_reply.started":"2022-01-10T22:31:53.734Z","shell.execute_reply":"2022-01-10T22:31:54.016045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the dimensions of the train metadata.\nprint('train_df dimensions: ', train_df.shape)\nprint('train_df column names: ', train_df.columns.values.tolist())\n\n#print an extra row could use '\\n' as well in a print statement\nprint('')\n\n#show the dimensions of the test metadata\nprint('test_df dimensions: ',test_df.shape)\nprint('test_df column names: ', test_df.columns.values.tolist())","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:54.018022Z","iopub.execute_input":"2022-01-10T22:31:54.018398Z","iopub.status.idle":"2022-01-10T22:31:54.027515Z","shell.execute_reply.started":"2022-01-10T22:31:54.018362Z","shell.execute_reply":"2022-01-10T22:31:54.026517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show the type of train_jpg and test_jpg as well as length of the list.\nprint('train_jpg is of type ',type(train_jpg), ' and length ', len(train_jpg))\n#Also show the first 3 elements\nprint('train_jpg list 1st 3 elements: ', train_jpg[0:3], '\\n')\n\nprint('test_jpg is of type ',type(test_jpg), ' and length ', len(test_jpg))\n#Also show the first 3 elements\nprint('test_jpg list 1st 3 elements: ', test_jpg[0:3])\n","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:54.030055Z","iopub.execute_input":"2022-01-10T22:31:54.032053Z","iopub.status.idle":"2022-01-10T22:31:54.044446Z","shell.execute_reply.started":"2022-01-10T22:31:54.032017Z","shell.execute_reply":"2022-01-10T22:31:54.043672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set(rc={'figure.figsize':(15,5)})\nfig = plt.figure()\nsns.histplot(data=train_df, x='Pawpularity', bins=100)\nplt.axvline(train_df['Pawpularity'].mean(), c='red', ls='-', lw=3, label='Mean Pawpularity')\nplt.axvline(train_df['Pawpularity'].median(),c='blue',ls='-',lw=3, label='Median Pawpularity')\nplt.title('Distribution of Pawpularity Scores', fontsize=20, fontweight='bold')\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:54.050119Z","iopub.execute_input":"2022-01-10T22:31:54.050571Z","iopub.status.idle":"2022-01-10T22:31:54.666833Z","shell.execute_reply.started":"2022-01-10T22:31:54.050535Z","shell.execute_reply":"2022-01-10T22:31:54.66616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's start with just one variable to demonstrate\nfig, ax = plt.subplots(1,2)\nsns.boxplot(data=train_df, x='Eyes', y='Pawpularity', ax=ax[0])\nsns.histplot(train_df, x=\"Pawpularity\", hue=\"Eyes\", kde=True, ax=ax[1])\nplt.suptitle(\"Eyes\", fontsize=20, fontweight='bold')\nfig.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:54.66805Z","iopub.execute_input":"2022-01-10T22:31:54.668368Z","iopub.status.idle":"2022-01-10T22:31:55.37829Z","shell.execute_reply.started":"2022-01-10T22:31:54.668331Z","shell.execute_reply":"2022-01-10T22:31:55.37762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#first let's try printing out the first 3 images with a loop\n\n#The for loop goes for 3 loops \nfor x in range(3):\n    #this loop goes through index of the train_jpg list of filenames: 0,1,2\n    image_path = train_jpg[x]\n    #use plt.imread() to read in that image file as an array of numbers between 0-255\n    image_array = plt.imread(image_path) \n    #Let's check the image dimensions\n    print(\"image {}'s dimensions are: {}\".format(x,image_array.shape))\n    #then plt.imshow() can display it for you\n    plt.imshow(image_array)\n    #title is the index of train_jpg\n    plt.title(x) \n    #turn off gridlines\n    plt.axis('off')\n    #show the image\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:55.379361Z","iopub.execute_input":"2022-01-10T22:31:55.379841Z","iopub.status.idle":"2022-01-10T22:31:55.997586Z","shell.execute_reply.started":"2022-01-10T22:31:55.379791Z","shell.execute_reply":"2022-01-10T22:31:55.996917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#first let's try printing out the first 3 images with a loop\n\n#The for loop goes for 3 loops \nfor x in range(8):\n    #this loop goes through index of the train_jpg list of filenames: 0,1,2\n    image_path = test_jpg[x]\n    #use plt.imread() to read in that image file as an array of numbers between 0-255\n    image_array = plt.imread(image_path) \n    #Let's check the image dimensions\n    print(\"image {}'s dimensions are: {}\".format(x,image_array.shape))\n    #then plt.imshow() can display it for you\n    plt.imshow(image_array)\n    #title is the index of train_jpg\n    plt.title(x) \n    #turn off gridlines\n    plt.axis('off')\n    #show the image\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:55.998781Z","iopub.execute_input":"2022-01-10T22:31:55.999113Z","iopub.status.idle":"2022-01-10T22:31:57.505375Z","shell.execute_reply.started":"2022-01-10T22:31:55.999077Z","shell.execute_reply":"2022-01-10T22:31:57.504733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE  \nimg_size = 224\nchannels = 3\nBatch_size = 128\n\n\n\ndef seed_everything():\n    np.random.seed(123)\n    random.seed(123)\n    tf.random.set_seed(123)\n    os.environ[\"TF_CPP_MIN_LOG_LEVEL\"] = '2'\n    os.environ['PYTHONHASHSEED'] = str(123)\n\nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:57.506601Z","iopub.execute_input":"2022-01-10T22:31:57.507099Z","iopub.status.idle":"2022-01-10T22:31:57.513571Z","shell.execute_reply.started":"2022-01-10T22:31:57.507061Z","shell.execute_reply":"2022-01-10T22:31:57.512888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### reading the data\ndf = pd.read_csv(\"/kaggle/input/petfinder-pawpularity-score-clean/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/petfinder-pawpularity-score/test.csv\")\nId = df_test[\"Id\"].copy()\n\n\ndf[\"Id\"] = df[\"Id\"].apply(lambda x : \"/kaggle/input/petfinder-pawpularity-score-clean/train/\" + x + \".jpg\")\ndf_test[\"Id\"] = df_test[\"Id\"].apply(lambda x : \"/kaggle/input/petfinder-pawpularity-score/test/\" + x + \".jpg\")","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:57.514786Z","iopub.execute_input":"2022-01-10T22:31:57.515537Z","iopub.status.idle":"2022-01-10T22:31:57.552746Z","shell.execute_reply.started":"2022-01-10T22:31:57.515501Z","shell.execute_reply":"2022-01-10T22:31:57.552076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Image preproccesing","metadata":{}},{"cell_type":"code","source":"### doing data augmentation in images\ndef image_preprocess(is_labelled):  \n    def augment(image):\n        image = tf.image.random_flip_left_right(image)\n\n        image = tf.image.random_saturation(image, 0.95, 1.05)\n        image = tf.image.random_contrast(image, 0.95, 1.05)\n        return image\n    \n    def can_be_augmented(img, label):\n        return augment(img), label\n    \n\n    return can_be_augmented if is_labelled else augment\n\n\n\n\ndef image_read(is_labelled):\n    def decode(path):\n        image = tf.io.read_file(path)\n        image = tf.image.decode_jpeg(image, channels=channels)\n        image = tf.cast(image, tf.float32)\n        image = tf.image.resize(image, (img_size, img_size))\n        image = tf.keras.applications.efficientnet.preprocess_input(image) \n        return image\n    \n    def can_be_decoded(path, label):\n        return decode(path), label\n    \n#   If record has label both image and lable will be returned\n\n    return can_be_decoded if is_labelled else decode\n\n\n\ndef create_dataset(df, batch_size, is_labelled = False, augment = False, shuffle = False):\n    image_read_fn = image_read(is_labelled)\n    image_preprocess_fn = image_preprocess(is_labelled)\n    \n    if is_labelled:\n        dataset = tf.data.Dataset.from_tensor_slices((df[\"Id\"].values, df[\"Pawpularity\"].values))\n    else:\n        dataset = tf.data.Dataset.from_tensor_slices((df[\"Id\"].values))\n    \n    dataset = dataset.map(image_read_fn, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.map(image_preprocess_fn, num_parallel_calls=AUTOTUNE) if augment else dataset\n    dataset = dataset.shuffle(1024, reshuffle_each_iteration=True) if shuffle else dataset\n    dataset = dataset.batch(batch_size)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:57.554568Z","iopub.execute_input":"2022-01-10T22:31:57.555074Z","iopub.status.idle":"2022-01-10T22:31:57.567562Z","shell.execute_reply.started":"2022-01-10T22:31:57.555035Z","shell.execute_reply":"2022-01-10T22:31:57.566854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:57.568855Z","iopub.execute_input":"2022-01-10T22:31:57.569149Z","iopub.status.idle":"2022-01-10T22:31:57.595445Z","shell.execute_reply.started":"2022-01-10T22:31:57.56909Z","shell.execute_reply":"2022-01-10T22:31:57.594768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn = df.iloc[:9000]\nval = df.iloc[9001:]\n\n\n\ntrain = create_dataset(trn, Batch_size, is_labelled = True, augment = True, shuffle = True)\nvalidation = create_dataset(val, Batch_size, is_labelled = True, augment = False, shuffle = False)\ntest = create_dataset(df_test, Batch_size, is_labelled = False, augment = False, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:31:57.596542Z","iopub.execute_input":"2022-01-10T22:31:57.596791Z","iopub.status.idle":"2022-01-10T22:32:00.220584Z","shell.execute_reply.started":"2022-01-10T22:31:57.596755Z","shell.execute_reply":"2022-01-10T22:32:00.219919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(trn.shape)\nprint(val.shape)\nprint(df_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:32:00.22356Z","iopub.execute_input":"2022-01-10T22:32:00.223799Z","iopub.status.idle":"2022-01-10T22:32:00.229306Z","shell.execute_reply.started":"2022-01-10T22:32:00.223765Z","shell.execute_reply":"2022-01-10T22:32:00.228624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_mod = \"/kaggle/input/keras-applications-models/EfficientNetB0.h5\"\nefnet = tf.keras.models.load_model(img_mod)\n\nefnet.trainable = False","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:32:00.230256Z","iopub.execute_input":"2022-01-10T22:32:00.230618Z","iopub.status.idle":"2022-01-10T22:32:02.674459Z","shell.execute_reply.started":"2022-01-10T22:32:00.230581Z","shell.execute_reply":"2022-01-10T22:32:02.673713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential([\n    Input(shape=(img_size, img_size, channels)),\n    efnet,\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(units = 128, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(units = 64, activation=\"relu\"),\n    BatchNormalization(),\n    Dropout(0.2),\n    Dense(units = 1, activation=\"relu\")\n])","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:32:02.675862Z","iopub.execute_input":"2022-01-10T22:32:02.676118Z","iopub.status.idle":"2022-01-10T22:32:03.312621Z","shell.execute_reply.started":"2022-01-10T22:32:02.676084Z","shell.execute_reply":"2022-01-10T22:32:03.311946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = EarlyStopping(patience = 5,restore_best_weights=True)\n\n\nlr_schedule = ExponentialDecay(\n    initial_learning_rate=1e-3,\n    decay_steps=100, decay_rate=0.96,\n    staircase=True)","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:32:03.313711Z","iopub.execute_input":"2022-01-10T22:32:03.315631Z","iopub.status.idle":"2022-01-10T22:32:03.319593Z","shell.execute_reply.started":"2022-01-10T22:32:03.3156Z","shell.execute_reply":"2022-01-10T22:32:03.318953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss=\"mse\", \n              optimizer = tf.keras.optimizers.Adam(learning_rate = lr_schedule), \n              metrics=[tf.keras.metrics.RootMeanSquaredError()])\n\npredictor = model.fit(train,\n                      epochs=50, \n                      validation_data = validation,\n                      callbacks=[early_stopping],\n                     use_multiprocessing=True, workers=-1)","metadata":{"execution":{"iopub.status.busy":"2022-01-10T22:49:55.656505Z","iopub.execute_input":"2022-01-10T22:49:55.656756Z","iopub.status.idle":"2022-01-10T23:13:19.158368Z","shell.execute_reply.started":"2022-01-10T22:49:55.656729Z","shell.execute_reply":"2022-01-10T23:13:19.157598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(test)\nfinal=pd.DataFrame()\nfinal['Id']=Id\nfinal['Pawpularity']=pred\nfinal.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-01-10T23:21:27.635729Z","iopub.execute_input":"2022-01-10T23:21:27.635999Z","iopub.status.idle":"2022-01-10T23:21:27.690094Z","shell.execute_reply.started":"2022-01-10T23:21:27.635966Z","shell.execute_reply":"2022-01-10T23:21:27.689339Z"},"trusted":true},"execution_count":null,"outputs":[]}]}