{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from pathlib import Path\nimport cv2\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom matplotlib import style\nstyle.use('fivethirtyeight')","metadata":{"execution":{"iopub.status.busy":"2022-03-09T15:45:55.732063Z","iopub.execute_input":"2022-03-09T15:45:55.732403Z","iopub.status.idle":"2022-03-09T15:45:56.086650Z","shell.execute_reply.started":"2022-03-09T15:45:55.732368Z","shell.execute_reply":"2022-03-09T15:45:56.085603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/ultra-mnist/test | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-03-09T15:38:55.745486Z","iopub.execute_input":"2022-03-09T15:38:55.746443Z","iopub.status.idle":"2022-03-09T15:38:56.594894Z","shell.execute_reply.started":"2022-03-09T15:38:55.746397Z","shell.execute_reply":"2022-03-09T15:38:56.594063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input/ultra-mnist/train | wc -l","metadata":{"execution":{"iopub.status.busy":"2022-03-09T15:39:12.221276Z","iopub.execute_input":"2022-03-09T15:39:12.221699Z","iopub.status.idle":"2022-03-09T15:39:13.559833Z","shell.execute_reply.started":"2022-03-09T15:39:12.221669Z","shell.execute_reply":"2022-03-09T15:39:13.558948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets have a look at labels\ndf = pd.read_csv('../input/ultra-mnist/train.csv')\ndf_test = pd.read_csv('../input/ultra-mnist/sample_submission.csv')\nvals, cnts = np.unique(df.digit_sum.values, return_counts=True)\nplt.title(f'Train: {len(df)} Test: {len(df_test)}')\nplt.barh(vals, cnts)\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-09T15:38:31.601396Z","iopub.execute_input":"2022-03-09T15:38:31.601709Z","iopub.status.idle":"2022-03-09T15:38:31.811787Z","shell.execute_reply.started":"2022-03-09T15:38:31.601663Z","shell.execute_reply":"2022-03-09T15:38:31.811151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# - Completely balaced train dataset & testset size is same as train set.\n# - Most probabably dataset is generated using some kind of sampling method (usually random)\n# - Expecting test dataset distribution also to be balanced! If not, organizers are foresighted and we have to focus on cv splits, else more on the pipeline","metadata":{"execution":{"iopub.status.busy":"2022-03-09T15:41:08.399449Z","iopub.execute_input":"2022-03-09T15:41:08.399784Z","iopub.status.idle":"2022-03-09T15:41:08.404145Z","shell.execute_reply.started":"2022-03-09T15:41:08.399748Z","shell.execute_reply":"2022-03-09T15:41:08.403248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"images = np.array(sorted(list(Path('../input/ultra-mnist/train').glob('*.jpeg'))))\nimages = images[np.random.randint(0, 100, size=100)]\nlabels = [df[df.id==p.stem].digit_sum.values[0] for p in images]\nf, axarr = plt.subplots(10,10, figsize=(25,25))\naxarr = axarr.flatten()\nfor i in range(100):\n    image = images[i]\n    label = labels[i]\n    im = plt.imread(str(image))\n    axarr[i].imshow(im)\n    axarr[i].set_title(f'sum: {label}')\n    axarr[i].set_xticks([])\n    axarr[i].set_yticks([])\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:01:22.718371Z","iopub.execute_input":"2022-03-09T16:01:22.718626Z","iopub.status.idle":"2022-03-09T16:02:20.527408Z","shell.execute_reply.started":"2022-03-09T16:01:22.718596Z","shell.execute_reply":"2022-03-09T16:02:20.522993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.hist(labels) # as sample (currently 100) size increase, more this plot will become uniform.","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:02:20.529012Z","iopub.execute_input":"2022-03-09T16:02:20.529872Z","iopub.status.idle":"2022-03-09T16:02:20.742497Z","shell.execute_reply.started":"2022-03-09T16:02:20.529798Z","shell.execute_reply":"2022-03-09T16:02:20.741614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}