{"cells":[{"metadata":{"_uuid":"9ea7e494dc99d4b5ac1c29ea3928929fdbbc078b"},"cell_type":"markdown","source":"This is my approach to predict mnist digit dataset, this isnt my best work, but i'll try to update it as im getting better at machine learning "},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt # plot \nfrom sklearn.model_selection import train_test_split \nfrom sklearn.metrics import confusion_matrix\n\n\n\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout\nfrom keras.utils import to_categorical\nfrom keras.optimizers import RMSprop\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"#read dataset\ntrain_set = pd.read_csv('../input/train.csv')\ntest_set = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20b9b911b12d008ab1a51d76139150d0d7b275e3"},"cell_type":"code","source":"print(train_set.shape)\nprint(test_set.shape) #one less column\ntrain_set.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bcf57d6725432091d406962bb01fedc641d4b329"},"cell_type":"code","source":"x_train = train_set.iloc[:,1:]\ny_train = train_set.loc[:,['label']]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bb35e3e36c2f40c6650ca49bb934d9736feb8694"},"cell_type":"code","source":"#Getting the train and test set\nX_train,X_test,y_train,y_test = train_test_split(x_train, y_train,test_size=0.3,shuffle = False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1e663be5146804a20e05dbdc4c3f3ecf816f426"},"cell_type":"code","source":"X_train/=255\nX_test/=255\ntest_set/=255","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f84bafc6a89ed1ec9e7cf8ed0c33b6f6e4653db8"},"cell_type":"code","source":"print(X_train.shape[0])\nprint(y_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d06061d54bcddca92d40cb76ee50bf91af6628f"},"cell_type":"code","source":"#for conv net\nX_train = X_train.values.reshape(X_train.shape[0],28,28,1)\nX_test = X_test.values.reshape(X_test.shape[0],28,28,1)\nval_test = test_set.values.reshape(test_set.shape[0],28,28,1)\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d8a5692578609ef007d1f14fe67d078d59a81ac6"},"cell_type":"code","source":"print(X_train.shape)\nprint(y_train.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b52d5f31a097a02255a9015660d19f33c37e49bc"},"cell_type":"code","source":"#visual = X_train.values.reshape(-1,28,28,1)\ng = plt.imshow(X_train[0][:,:,0])\nprint(y_train.values[0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"942b1896d64834cfbfcbf856654d91a16daba6dc"},"cell_type":"code","source":"#one hot encoder\ny_train =  to_categorical(y_train, 10)\ny_test =  to_categorical(y_test, 10)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42bf453d173119756da399af2de4f075fe95c5ae"},"cell_type":"code","source":"#image argumentation\nfrom keras.preprocessing.image import ImageDataGenerator\nX_train2 = np.array(X_train, copy=True) \ny_train2 = np.array(y_train, copy=True) \n\ndatagen = ImageDataGenerator(\n    featurewise_center=True,\n    featurewise_std_normalization=True,\n    rotation_range=20,\n    )\n\ndatagen.fit(X_train)\n\nprint(type(X_train2))\nprint(type(X_train))\n\n# Concatenating\nresult_x  = np.concatenate((X_train, X_train2), axis=0)\nresult_y  = np.concatenate((y_train, y_train2), axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"01967b3ba8abc661cd3e5fa89b4d3c395d55a036"},"cell_type":"code","source":"print(X_train.shape)\nprint(result_x.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5f6548e7d95e92b901faf8f5c02ec1fb7206e9ab"},"cell_type":"code","source":"def mlp_nn():\n    model = Sequential()\n    model.add(Dense(64 ,activation='relu', input_dim =784))\n    model.add(Dense(10,activation='softmax'))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3999a04eb6715c209b57d50a723521d19144946c"},"cell_type":"code","source":"def conv_nn():\n    from keras.layers import Conv2D, MaxPooling2D, Flatten\n    model = Sequential()\n    model.add(Conv2D(32,(3,3),activation='relu', input_shape =(28,28,1)))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Dropout(0,4))\n    model.add(Conv2D(64,(3,3),activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2,2)))\n    model.add(Dropout(0,4))\n    model.add(Flatten())\n    model.add(Dense(128 ,activation='relu'))\n    model.add(Dropout(0,5))\n    model.add(Dense(128 ,activation='relu'))\n    model.add(Dropout(0,5))\n    model.add(Dense(10,activation='softmax'))\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4468a7e90ac3d628e83bb2e2081435fd28c4b2e7"},"cell_type":"code","source":"model = conv_nn()\nmodel.compile(optimizer=RMSprop(),\n loss='categorical_crossentropy',\n metrics=['accuracy'])\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3d38eaa0b5c45c99bebba70a54d6f808b00aa8a2"},"cell_type":"code","source":"model.fit(result_x,result_y,epochs=12,batch_size=128,validation_data=(X_test,y_test))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9665edde229195851fa25400e2e380176efb9108"},"cell_type":"code","source":"prediction = model.evaluate(X_test,y_test,batch_size=32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2bf5c151b3e0a425ec3a4f1042bd793b882dd4f3"},"cell_type":"code","source":"model.predict_classes(val_test, batch_size=32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39ac489f8e8e3c0bbda5e79a3f63543c3a1b36cb"},"cell_type":"code","source":"g2 = plt.imshow(val_test[-3][:,:,0])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3dff36718076367fc50c2ecd3d5c297cb74f0192"},"cell_type":"code","source":"def plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=45)\n    plt.yticks(tick_marks, classes)\n\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e9a058b361c944301a60b654ff9d11b4ac184ee"},"cell_type":"code","source":"end_file = model.predict_classes(val_test, batch_size=32)\nresults = pd.Series(end_file,name=\"Label\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"badcf8f48dde4e15fe00c2dfefb4073b091b0873"},"cell_type":"code","source":"submission = pd.concat([pd.Series(range(1,28001),name = \"ImageId\"),results],axis = 1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b79edec3c8b9c38316c3e1e683b79b12259d7a1f"},"cell_type":"code","source":"submission.to_csv(\"mnist_dense.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6565732c92f564fc227070c27d371991b3bbeb88"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"794affa66a5fd1c687641a17f2815051ce391344"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}