{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport colorama\nimport tensorflow as tf\n\nfrom tensorflow.keras import datasets, layers, models\nfrom keras.utils.np_utils import to_categorical\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, Dense, MaxPooling2D, Activation, Dropout, Flatten\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames: \n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the  directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \nprint(\"Completed\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T12:02:09.889554Z","iopub.execute_input":"2022-08-10T12:02:09.889967Z","iopub.status.idle":"2022-08-10T12:02:17.114273Z","shell.execute_reply.started":"2022-08-10T12:02:09.889858Z","shell.execute_reply":"2022-08-10T12:02:17.113017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Import the dataset\nEN\nWe use pandas to import the csv file\nAnd .head to see a piece of the data\n\nES\nUsamos pandas para importar el archivo csv\nY .head para ver una parte de los datos","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/digit-recognizer/train.csv\")\ntest_data = pd.read_csv(\"../input/digit-recognizer/test.csv\")\n\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:17.116453Z","iopub.execute_input":"2022-08-10T12:02:17.117495Z","iopub.status.idle":"2022-08-10T12:02:22.119862Z","shell.execute_reply.started":"2022-08-10T12:02:17.117457Z","shell.execute_reply":"2022-08-10T12:02:22.118825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EN\nFunction that we use later to see the characteristics of the data such as missing values (NaN), all of features and number, records and columns\n\nES\nFunción que usamos más tarde para ver las características de los datos, como valores faltantes (NaN), todas las características y el número, registros y columnas.","metadata":{}},{"cell_type":"code","source":"def data_description(df):\n    print(\"Data description\")\n    print(f\"Total number of records {df.shape[0]}\")\n    print(f'number of features {df.shape[1]}\\n\\n')\n    columns = df.columns\n    data_type = []\n    \n    # Get the datatype of features\n    for col in df.columns:\n        data_type.append(df[col].dtype)\n        \n    n_uni = df.nunique()\n    # Number of NaN values\n    n_miss = df.isna().sum()\n    \n    names = list(zip(columns, data_type, n_uni, n_miss))\n    variable_desc = pd.DataFrame(names, columns=[\"Name\",\"Type\",\"Unique levels\",\"Missing\"])\n    print(variable_desc)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:22.121500Z","iopub.execute_input":"2022-08-10T12:02:22.122168Z","iopub.status.idle":"2022-08-10T12:02:22.129751Z","shell.execute_reply.started":"2022-08-10T12:02:22.122130Z","shell.execute_reply":"2022-08-10T12:02:22.128508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_description(train_data) # Show information of train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:22.132460Z","iopub.execute_input":"2022-08-10T12:02:22.132939Z","iopub.status.idle":"2022-08-10T12:02:22.440052Z","shell.execute_reply.started":"2022-08-10T12:02:22.132906Z","shell.execute_reply":"2022-08-10T12:02:22.438770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_description(test_data) # Show information of test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:22.441972Z","iopub.execute_input":"2022-08-10T12:02:22.442370Z","iopub.status.idle":"2022-08-10T12:02:22.677811Z","shell.execute_reply.started":"2022-08-10T12:02:22.442334Z","shell.execute_reply":"2022-08-10T12:02:22.676936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Drop label from train values\nAnd set in Y label values","metadata":{}},{"cell_type":"code","source":"X = train_data.drop(\"label\", axis = 1)\ny = train_data[\"label\"]\nprint(\"Completed\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:22.679469Z","iopub.execute_input":"2022-08-10T12:02:22.679821Z","iopub.status.idle":"2022-08-10T12:02:22.765939Z","shell.execute_reply.started":"2022-08-10T12:02:22.679786Z","shell.execute_reply":"2022-08-10T12:02:22.764816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.value_counts() # Count all values with the same index ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:22.767598Z","iopub.execute_input":"2022-08-10T12:02:22.768302Z","iopub.status.idle":"2022-08-10T12:02:22.778251Z","shell.execute_reply.started":"2022-08-10T12:02:22.768263Z","shell.execute_reply":"2022-08-10T12:02:22.777047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's show information related witn index number","metadata":{}},{"cell_type":"code","source":"result = y.value_counts()\nplt.figure(figsize = (10,5))\nplt.title(\"Number of items in each category\")\nplt.ylabel(\"Number (item)\")\nresult.plot.bar(color = sns.color_palette(\"Spectral\",10), width = 0.8)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:22.780283Z","iopub.execute_input":"2022-08-10T12:02:22.780683Z","iopub.status.idle":"2022-08-10T12:02:23.026889Z","shell.execute_reply.started":"2022-08-10T12:02:22.780651Z","shell.execute_reply":"2022-08-10T12:02:23.025741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EN**\nWe verify that there is no great difference between the different values ​​that could alter the result of our model.\n\n**ES**\nComprobamos que no existe gran diferencia entre los distintos valores que pudiera alterar el resultado de nuestro modelo.","metadata":{}},{"cell_type":"markdown","source":"**EN** Let's check the number of pixels each image has with .shape()\n**ES** Comprobemos la cantidad de píxeles que tiene cada imagen con .shape()","metadata":{}},{"cell_type":"code","source":"X.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:23.028525Z","iopub.execute_input":"2022-08-10T12:02:23.029176Z","iopub.status.idle":"2022-08-10T12:02:23.035703Z","shell.execute_reply.started":"2022-08-10T12:02:23.029138Z","shell.execute_reply":"2022-08-10T12:02:23.034760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EN**\nThe number of pixels per image is 784, to get the width and height, we are going to do the square root\n**ES**\nEl número de píxeles por imagen es 784, para obtener el ancho y el alto, vamos a hacer la raíz cuadrada","metadata":{}},{"cell_type":"code","source":"w_and_h_value = X.shape[1] ** 0.5 # Do the square root to get the width and height value\nprint(f'Value to use in (-1, ?, ?, 1) {w_and_h_value}')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:23.040119Z","iopub.execute_input":"2022-08-10T12:02:23.040744Z","iopub.status.idle":"2022-08-10T12:02:23.046268Z","shell.execute_reply.started":"2022-08-10T12:02:23.040708Z","shell.execute_reply":"2022-08-10T12:02:23.045235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.values.reshape(-1, 28, 28, 1) # 1 = gray color scale\nX_test = test_data.values.reshape(-1,28,28,1)\n\nX.shape, X_test.shape # (Number of values, width, height, gray scale)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:23.047777Z","iopub.execute_input":"2022-08-10T12:02:23.048419Z","iopub.status.idle":"2022-08-10T12:02:23.060659Z","shell.execute_reply.started":"2022-08-10T12:02:23.048376Z","shell.execute_reply":"2022-08-10T12:02:23.059462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.shape[0] # Number of images","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:23.062173Z","iopub.execute_input":"2022-08-10T12:02:23.062655Z","iopub.status.idle":"2022-08-10T12:02:23.071968Z","shell.execute_reply.started":"2022-08-10T12:02:23.062620Z","shell.execute_reply":"2022-08-10T12:02:23.071060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's show random images","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(15,10)) # setting figure size\nfor i in range(10): # running loop 10 times to print 10 digit at once\n    plt.subplot(5,5,i+1)\n    index = np.random.randint(0,42000) # picaing random image from our whole dataset\n    plt.imshow(X[index],cmap='Greys') # printing picture in black and white format using cmap\n    plt.title(y[index]) # assigning the labeles to the pictures \n    plt.axis(False) # removing x axis and y axis bars from image \n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:23.073410Z","iopub.execute_input":"2022-08-10T12:02:23.073745Z","iopub.status.idle":"2022-08-10T12:02:23.500780Z","shell.execute_reply.started":"2022-08-10T12:02:23.073713Z","shell.execute_reply":"2022-08-10T12:02:23.499796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Add data augmentation\n**EN**\nTo try to get more data to train the model, we modify the original values and make some \"fake\" values to increase the reliability it offers us.\n\n**ES**\nPara intentar sacar más datos para entrenar el modelo modificamos los valores originales y hacemos unos valores \"falsos\" para aumentar la confianza que nos ofrece.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.image import ImageDataGenerator\n\ndatagen = ImageDataGenerator(\n    rotation_range = 20, # maximum level of rotation\n    shear_range = 30,# \n    brightness_range=[0, 5],# The range of brightness that the can increase\n)\n\ndatagen.fit(X) # Get the modified data\n\n# Show the modified data\nplt.figure(figsize = (20, 8))\nfor img, label in datagen.flow(X, y, batch_size = 10, shuffle = False):\n    for i in range(10):\n        plt.subplot(2, 5, i+1)\n        plt.yticks([])\n        plt.imshow(img[i].reshape(28, 28), cmap = \"gray\")\n    break","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:23.502456Z","iopub.execute_input":"2022-08-10T12:02:23.502810Z","iopub.status.idle":"2022-08-10T12:02:24.327459Z","shell.execute_reply.started":"2022-08-10T12:02:23.502775Z","shell.execute_reply":"2022-08-10T12:02:24.326374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data normalization","metadata":{}},{"cell_type":"markdown","source":"**EN**\nTransform the data, to have them between values between 0 and 1, this is because the white color has the minimum value (0) and the black color the maximum (255)\n\n**ES**\nTransforma los datos, para tenerlos entre valores entre 0 y 1, esto es porque el color blanco tiene el valor mínimo (0) y el negro el máximo (255)","metadata":{}},{"cell_type":"code","source":"X = X/255\nX_test = X_test / 255","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:24.329118Z","iopub.execute_input":"2022-08-10T12:02:24.329778Z","iopub.status.idle":"2022-08-10T12:02:24.503934Z","shell.execute_reply.started":"2022-08-10T12:02:24.329740Z","shell.execute_reply":"2022-08-10T12:02:24.502801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EN**\nConvert the y to categorical\nnum_classes = all categories, in this case (0 to 9) \n\n\n**ES**\nConvierte la y en categórica\nnum_classes = todas las categorías en este caso (0 a 9)","metadata":{}},{"cell_type":"code","source":"y = to_categorical(y, num_classes = 10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:24.505751Z","iopub.execute_input":"2022-08-10T12:02:24.506145Z","iopub.status.idle":"2022-08-10T12:02:24.511684Z","shell.execute_reply.started":"2022-08-10T12:02:24.506107Z","shell.execute_reply":"2022-08-10T12:02:24.510727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EN**\nGet training and validation data\n\n**ES**\nObtiene los datos de entrenamiento y validación","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size = 0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:24.513182Z","iopub.execute_input":"2022-08-10T12:02:24.513818Z","iopub.status.idle":"2022-08-10T12:02:24.995169Z","shell.execute_reply.started":"2022-08-10T12:02:24.513780Z","shell.execute_reply":"2022-08-10T12:02:24.994139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Create the model","metadata":{}},{"cell_type":"code","source":"model = tf.keras.models.Sequential([\n    tf.keras.layers.Conv2D(32, (3,3), activation = \"relu\", input_shape=(28,28,1)), # input_shape = dimension of the image and color\n    tf.keras.layers.MaxPooling2D(2,2), # Downsamples the input along its spatial dimensions (height and width) by taking the maximum value over an input window\n    tf.keras.layers.Dropout(0.2), # Drop a percent of neurons\n    \n    tf.keras.layers.Conv2D(64, (3,3), activation = \"relu\"),\n    tf.keras.layers.MaxPooling2D(2,2),\n\n    tf.keras.layers.Conv2D(128, (3,3), activation = \"relu\"),\n    tf.keras.layers.MaxPooling2D(2,2),\n    tf.keras.layers.Dropout(0.2),\n    \n    tf.keras.layers.Flatten(), # eshaping it into a one-dimensional tensor.\n    \n    tf.keras.layers.Dense(125, activation = \"relu\"),\n    tf.keras.layers.Dense(10, activation = \"softmax\"), # Output neuron\n])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:24.996513Z","iopub.execute_input":"2022-08-10T12:02:24.997005Z","iopub.status.idle":"2022-08-10T12:02:27.916712Z","shell.execute_reply.started":"2022-08-10T12:02:24.996968Z","shell.execute_reply":"2022-08-10T12:02:27.915767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compile the model","metadata":{}},{"cell_type":"code","source":"model.compile(optimizer = 'adam',\n              loss='categorical_crossentropy',\n              metrics=['accuracy']\n              )","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:27.918161Z","iopub.execute_input":"2022-08-10T12:02:27.918530Z","iopub.status.idle":"2022-08-10T12:02:28.118942Z","shell.execute_reply.started":"2022-08-10T12:02:27.918494Z","shell.execute_reply":"2022-08-10T12:02:28.117939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Fit the model","metadata":{}},{"cell_type":"code","source":"data_gen_training = datagen.flow(X_train, y_train, batch_size = 28)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:28.120490Z","iopub.execute_input":"2022-08-10T12:02:28.120872Z","iopub.status.idle":"2022-08-10T12:02:28.163307Z","shell.execute_reply.started":"2022-08-10T12:02:28.120836Z","shell.execute_reply":"2022-08-10T12:02:28.162269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(data_gen_training,\n             validation_data = (X_val, y_val),\n             epochs = 75, batch_size = 32, # Epochs number of times to repet the proces\n             )","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:02:28.164893Z","iopub.execute_input":"2022-08-10T12:02:28.165286Z","iopub.status.idle":"2022-08-10T12:22:32.052367Z","shell.execute_reply.started":"2022-08-10T12:02:28.165249Z","shell.execute_reply":"2022-08-10T12:22:32.051298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_frame = pd.DataFrame(history.history)\nhistory_frame.loc[:, ['loss', 'val_loss']].plot()\nhistory_frame.loc[:, ['accuracy', 'val_accuracy']].plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:22:32.053959Z","iopub.execute_input":"2022-08-10T12:22:32.054535Z","iopub.status.idle":"2022-08-10T12:22:32.418841Z","shell.execute_reply.started":"2022-08-10T12:22:32.054497Z","shell.execute_reply":"2022-08-10T12:22:32.417807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model.predict(X_test)\npred = tf.math.argmax(pred, axis = -1)\npred = pd.Series(pred, name='Label')\n\nimage_id = pd.Series(range(1,28001),name='ImageId')\nimage_id.isnull().sum()\n\npred = pd.concat([image_id,pred],axis=1)\npred.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T12:22:32.420683Z","iopub.execute_input":"2022-08-10T12:22:32.421720Z","iopub.status.idle":"2022-08-10T12:22:34.046253Z","shell.execute_reply.started":"2022-08-10T12:22:32.421681Z","shell.execute_reply":"2022-08-10T12:22:34.045265Z"},"trusted":true},"execution_count":null,"outputs":[]}]}