{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-30T23:21:23.837187Z","iopub.execute_input":"2022-07-30T23:21:23.837648Z","iopub.status.idle":"2022-07-30T23:21:23.869359Z","shell.execute_reply.started":"2022-07-30T23:21:23.837561Z","shell.execute_reply":"2022-07-30T23:21:23.868143Z"}}},{"cell_type":"code","source":"base_dir = \"/kaggle/input/dogs-vs-cats/\"\ntrain_dir = os.path.join(base_dir, \"train.zip\")\ntest_dir = os.path.join(base_dir, \"test1.zip\")\n\nimport zipfile\nwith zipfile.ZipFile(train_dir,\"r\") as z:\n    z.extractall()\n\nwith zipfile.ZipFile(test_dir,\"r\") as z:\n    z.extractall()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:21:23.871925Z","iopub.execute_input":"2022-07-30T23:21:23.872384Z","iopub.status.idle":"2022-07-30T23:21:43.099142Z","shell.execute_reply.started":"2022-07-30T23:21:23.872343Z","shell.execute_reply":"2022-07-30T23:21:43.098254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = os.listdir(\"./train\")\ndata = pd.DataFrame(images)\ndata = data.rename(columns={0: \"image\"})\ndata['image'] = data['image'].apply(lambda x: \"./train/\"+x)\ndata['label'] = data['image'].apply(lambda x: 0 if 'cat' in x else 1)\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:21:43.100426Z","iopub.execute_input":"2022-07-30T23:21:43.101229Z","iopub.status.idle":"2022-07-30T23:21:43.160360Z","shell.execute_reply.started":"2022-07-30T23:21:43.101183Z","shell.execute_reply":"2022-07-30T23:21:43.159315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D\nfrom keras.preprocessing import image\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.utils import to_categorical\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:21:43.162569Z","iopub.execute_input":"2022-07-30T23:21:43.162885Z","iopub.status.idle":"2022-07-30T23:21:50.913529Z","shell.execute_reply.started":"2022-07-30T23:21:43.162850Z","shell.execute_reply":"2022-07-30T23:21:50.912591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have grayscale images, so while loading the images we will keep grayscale=True, if you have RGB images, you should set grayscale as False\ntrain_image = []\nfor i in tqdm(range(data.shape[0])):\n    img = image.load_img(data['image'][i], target_size=(64,64,1), grayscale=True)\n    img = image.img_to_array(img)\n    img = img/255\n    train_image.append(img)\nX = np.array(train_image)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:21:50.916304Z","iopub.execute_input":"2022-07-30T23:21:50.917862Z","iopub.status.idle":"2022-07-30T23:23:10.272217Z","shell.execute_reply.started":"2022-07-30T23:21:50.917811Z","shell.execute_reply":"2022-07-30T23:23:10.270518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y=data['label'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:23:10.274059Z","iopub.execute_input":"2022-07-30T23:23:10.274438Z","iopub.status.idle":"2022-07-30T23:23:10.282400Z","shell.execute_reply.started":"2022-07-30T23:23:10.274403Z","shell.execute_reply":"2022-07-30T23:23:10.281024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X, y, random_state=42, test_size=0.2)","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:23:10.284461Z","iopub.execute_input":"2022-07-30T23:23:10.284975Z","iopub.status.idle":"2022-07-30T23:23:10.520137Z","shell.execute_reply.started":"2022-07-30T23:23:10.284933Z","shell.execute_reply":"2022-07-30T23:23:10.519034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential()\nmodel.add(Conv2D(32, kernel_size=(3,3), input_shape=(64, 64, 1), activation='relu'))\nmodel.add(Conv2D(64, kernel_size=(3,3), activation = 'relu'))\nmodel.add(MaxPooling2D(pool_size=2))\nmodel.add(Dropout(0.25))\nmodel.add(Flatten())\nmodel.add(Dense(128, activation='relu'))\nmodel.add(Dropout(0.5))\nmodel.add(Dense(1, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:23:10.521639Z","iopub.execute_input":"2022-07-30T23:23:10.522307Z","iopub.status.idle":"2022-07-30T23:23:10.784781Z","shell.execute_reply.started":"2022-07-30T23:23:10.522262Z","shell.execute_reply":"2022-07-30T23:23:10.783835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:23:10.790066Z","iopub.execute_input":"2022-07-30T23:23:10.790750Z","iopub.status.idle":"2022-07-30T23:23:10.797325Z","shell.execute_reply.started":"2022-07-30T23:23:10.790707Z","shell.execute_reply":"2022-07-30T23:23:10.796135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(loss='binary_crossentropy',optimizer='Adam',metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:23:10.799028Z","iopub.execute_input":"2022-07-30T23:23:10.799587Z","iopub.status.idle":"2022-07-30T23:23:10.821331Z","shell.execute_reply.started":"2022-07-30T23:23:10.799542Z","shell.execute_reply":"2022-07-30T23:23:10.819980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(X_train, y_train, epochs=20, validation_data=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-30T23:23:10.823084Z","iopub.execute_input":"2022-07-30T23:23:10.823894Z","iopub.status.idle":"2022-07-31T00:00:35.236781Z","shell.execute_reply.started":"2022-07-30T23:23:10.823847Z","shell.execute_reply":"2022-07-31T00:00:35.235018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n# summarize history for accuracy\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.title('model accuracy')\nplt.ylabel('accuracy')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:00:35.239200Z","iopub.execute_input":"2022-07-31T00:00:35.239598Z","iopub.status.idle":"2022-07-31T00:00:35.517881Z","shell.execute_reply.started":"2022-07-31T00:00:35.239565Z","shell.execute_reply":"2022-07-31T00:00:35.516537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# summarize history for loss\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title('model loss')\nplt.ylabel('loss')\nplt.xlabel('epoch')\nplt.legend(['train', 'test'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:00:35.519923Z","iopub.execute_input":"2022-07-31T00:00:35.520315Z","iopub.status.idle":"2022-07-31T00:00:35.727328Z","shell.execute_reply.started":"2022-07-31T00:00:35.520284Z","shell.execute_reply":"2022-07-31T00:00:35.726060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = os.listdir(\"./test1\")\ndf = pd.DataFrame(test)\n\ndf = df.rename(columns={0: \"image\"})\ndf['id'] = df['image'].apply(lambda x: x.split('.')[0])\ndf['image'] = df['image'].apply(lambda x: \"./test1/\"+x)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:12:50.696038Z","iopub.execute_input":"2022-07-31T00:12:50.696545Z","iopub.status.idle":"2022-07-31T00:12:50.740530Z","shell.execute_reply.started":"2022-07-31T00:12:50.696510Z","shell.execute_reply":"2022-07-31T00:12:50.739148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We have grayscale images, so while loading the images we will keep grayscale=True, if you have RGB images, you should set grayscale as False\nval_image = []\nfor i in tqdm(range(df.shape[0])):\n    img = image.load_img(df['image'][i], target_size=(64,64,1), grayscale=True)\n    img = image.img_to_array(img)\n    img = img/255\n    val_image.append(img)\nX_val = np.array(val_image)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:12:55.774985Z","iopub.execute_input":"2022-07-31T00:12:55.775368Z","iopub.status.idle":"2022-07-31T00:13:33.956136Z","shell.execute_reply.started":"2022-07-31T00:12:55.775337Z","shell.execute_reply":"2022-07-31T00:13:33.955076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(X_val)\ny_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:14:19.520047Z","iopub.execute_input":"2022-07-31T00:14:19.520501Z","iopub.status.idle":"2022-07-31T00:14:34.915083Z","shell.execute_reply.started":"2022-07-31T00:14:19.520466Z","shell.execute_reply":"2022-07-31T00:14:34.914262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['label']=y_pred\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:15:47.577299Z","iopub.execute_input":"2022-07-31T00:15:47.577712Z","iopub.status.idle":"2022-07-31T00:15:47.589846Z","shell.execute_reply.started":"2022-07-31T00:15:47.577682Z","shell.execute_reply":"2022-07-31T00:15:47.588606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['label'] = df['label'].apply(lambda x: 0 if x<0.5 else 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:15:51.735224Z","iopub.execute_input":"2022-07-31T00:15:51.735614Z","iopub.status.idle":"2022-07-31T00:15:51.750012Z","shell.execute_reply.started":"2022-07-31T00:15:51.735584Z","shell.execute_reply":"2022-07-31T00:15:51.748558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission = pd.DataFrame({'id': df.id, 'label': df.label})\n# you could use any filename. We choose submission here\nmy_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:15:53.434712Z","iopub.execute_input":"2022-07-31T00:15:53.435148Z","iopub.status.idle":"2022-07-31T00:15:53.458378Z","shell.execute_reply.started":"2022-07-31T00:15:53.435098Z","shell.execute_reply":"2022-07-31T00:15:53.457449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-31T00:15:55.095726Z","iopub.execute_input":"2022-07-31T00:15:55.096177Z","iopub.status.idle":"2022-07-31T00:15:55.116609Z","shell.execute_reply.started":"2022-07-31T00:15:55.096138Z","shell.execute_reply":"2022-07-31T00:15:55.115313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}