{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom PIL import Image\nimport plotly.express as px\nimport matplotlib.pyplot as plt","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-09T18:28:56.962386Z","iopub.execute_input":"2022-03-09T18:28:56.963211Z","iopub.status.idle":"2022-03-09T18:28:58.659468Z","shell.execute_reply.started":"2022-03-09T18:28:56.962969Z","shell.execute_reply":"2022-03-09T18:28:58.658198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/ultra-mnist/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:28:58.662304Z","iopub.execute_input":"2022-03-09T18:28:58.662655Z","iopub.status.idle":"2022-03-09T18:28:58.704527Z","shell.execute_reply.started":"2022-03-09T18:28:58.662601Z","shell.execute_reply":"2022-03-09T18:28:58.703454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:28:58.705939Z","iopub.execute_input":"2022-03-09T18:28:58.706247Z","iopub.status.idle":"2022-03-09T18:28:58.729835Z","shell.execute_reply.started":"2022-03-09T18:28:58.706211Z","shell.execute_reply":"2022-03-09T18:28:58.728865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataset Distribution\nLet us look at how the training data is distributed","metadata":{}},{"cell_type":"code","source":"px.bar(x=train['digit_sum'].value_counts().index, y=train['digit_sum'].value_counts().values, color=train['digit_sum'].value_counts().index)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:28:58.732449Z","iopub.execute_input":"2022-03-09T18:28:58.733369Z","iopub.status.idle":"2022-03-09T18:28:59.850203Z","shell.execute_reply.started":"2022-03-09T18:28:58.733313Z","shell.execute_reply":"2022-03-09T18:28:59.849197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- We see that the sums are equally distributed","metadata":{}},{"cell_type":"markdown","source":"## Visualize some sample images","metadata":{"execution":{"iopub.status.busy":"2022-03-09T17:27:46.142342Z","iopub.execute_input":"2022-03-09T17:27:46.142766Z","iopub.status.idle":"2022-03-09T17:27:46.154968Z","shell.execute_reply.started":"2022-03-09T17:27:46.142734Z","shell.execute_reply":"2022-03-09T17:27:46.153874Z"}}},{"cell_type":"code","source":"idx = 25\nimg = Image.open(f\"../input/ultra-mnist/train/{train['id'][idx]}.jpeg\")\nlabel = train['digit_sum'][idx]\nplt.figure(figsize=(15,15))\nplt.imshow(img)\nplt.title(f\"Sum = {label}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:28:59.851929Z","iopub.execute_input":"2022-03-09T18:28:59.852301Z","iopub.status.idle":"2022-03-09T18:29:02.880873Z","shell.execute_reply.started":"2022-03-09T18:28:59.852252Z","shell.execute_reply":"2022-03-09T18:29:02.879917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's look at the smallest digit in the sample image","metadata":{}},{"cell_type":"code","source":"plt.imshow(np.array(img)[1860:1910, 2700:2730])","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:29:02.882573Z","iopub.execute_input":"2022-03-09T18:29:02.882917Z","iopub.status.idle":"2022-03-09T18:29:03.237903Z","shell.execute_reply.started":"2022-03-09T18:29:02.882871Z","shell.execute_reply":"2022-03-09T18:29:03.237027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- The smallest sized digit in the above image is approximately of size 40x30 pixels","metadata":{}},{"cell_type":"markdown","source":"### Sample of a confusing perspective","metadata":{}},{"cell_type":"code","source":"idx = 1234\nimg = Image.open(f\"../input/ultra-mnist/train/{train['id'][idx]}.jpeg\")\nlabel = train['digit_sum'][idx]\nplt.figure(figsize=(15,15))\nplt.imshow(img)\nplt.title(f\"Sum = {label}\")","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-03-09T18:29:03.239691Z","iopub.execute_input":"2022-03-09T18:29:03.240451Z","iopub.status.idle":"2022-03-09T18:29:06.231438Z","shell.execute_reply.started":"2022-03-09T18:29:03.240385Z","shell.execute_reply":"2022-03-09T18:29:06.230543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- This image is a bit slightly confusing, whether to take the three perfectly perpendicular lines as 3 1's or just a single one. In this case it sums up to 6, so, we might have to consider it as single 1","metadata":{}},{"cell_type":"markdown","source":"## Samples of different sum values from 0 to 27","metadata":{}},{"cell_type":"code","source":"def create_subplot(idx, ax=None):\n    #https://stackoverflow.com/a/20073551\n    if ax is None:\n        ax = plt.gca()\n    img = Image.open(f\"../input/ultra-mnist/train/{train['id'][idx]}.jpeg\")\n    subplot = ax.imshow(img, cmap='gray')\n    ax.set_title(f\"Sum = {train['digit_sum'][idx]}\")\n    ax.tick_params(left=False, bottom=False, labelleft=False, labelbottom=False)\n    return subplot","metadata":{"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2022-03-09T18:29:06.233411Z","iopub.execute_input":"2022-03-09T18:29:06.233987Z","iopub.status.idle":"2022-03-09T18:29:06.241819Z","shell.execute_reply.started":"2022-03-09T18:29:06.233939Z","shell.execute_reply":"2022-03-09T18:29:06.240898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"no_of_samples = 6\nsums = 28 \n\nfig, axs = plt.subplots(sums,no_of_samples, figsize=(4*no_of_samples, 4*sums))\n\nfor i in range(sums):\n    indices = np.random.choice(train[train['digit_sum']==i].index.values, replace=False, size=no_of_samples)\n    for j, idx in enumerate(indices):\n        create_subplot(idx, axs[i, j])","metadata":{"execution":{"iopub.status.busy":"2022-03-09T18:29:06.243260Z","iopub.execute_input":"2022-03-09T18:29:06.243616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's look at some random samples","metadata":{}},{"cell_type":"code","source":"img = Image.open(f\"../input/ultra-mnist/train/{train['id'][4]}.jpeg\")\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}