{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\n","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:09:52.880702Z","iopub.execute_input":"2022-03-09T16:09:52.880969Z","iopub.status.idle":"2022-03-09T16:09:52.885175Z","shell.execute_reply.started":"2022-03-09T16:09:52.880940Z","shell.execute_reply":"2022-03-09T16:09:52.884602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/ultra-mnist/train.csv')\nsubmission = pd.read_csv('../input/ultra-mnist/sample_submission.csv')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:09:53.081644Z","iopub.execute_input":"2022-03-09T16:09:53.082315Z","iopub.status.idle":"2022-03-09T16:09:53.139640Z","shell.execute_reply.started":"2022-03-09T16:09:53.082270Z","shell.execute_reply":"2022-03-09T16:09:53.138761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:09:53.229286Z","iopub.execute_input":"2022-03-09T16:09:53.230069Z","iopub.status.idle":"2022-03-09T16:09:53.236299Z","shell.execute_reply.started":"2022-03-09T16:09:53.230017Z","shell.execute_reply":"2022-03-09T16:09:53.235582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_bins = 28\nsubmission['digit_sum'] = [28000-x for x in range(28000)]\nsubmission[\"digit_sum\"] = pd.cut(submission[\"digit_sum\"], bins=num_bins, labels=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:12:20.637606Z","iopub.execute_input":"2022-03-09T16:12:20.638360Z","iopub.status.idle":"2022-03-09T16:12:20.659759Z","shell.execute_reply.started":"2022-03-09T16:12:20.638322Z","shell.execute_reply":"2022-03-09T16:12:20.658985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('beat_bojan.csv', index=False)\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-09T16:12:21.474037Z","iopub.execute_input":"2022-03-09T16:12:21.474781Z","iopub.status.idle":"2022-03-09T16:12:21.538134Z","shell.execute_reply.started":"2022-03-09T16:12:21.474739Z","shell.execute_reply":"2022-03-09T16:12:21.537322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}