{"cells":[{"metadata":{},"cell_type":"markdown","source":"This is just a typing practice\nSource: https://www.kaggle.com/yuliagm/how-to-work-with-big-datasets-on-16g-ram-dask"},{"metadata":{"trusted":true},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport datetime\nimport os\nimport time\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport gc\n%matplotlib inline","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#make wider graphs\nsns.set(rc = {'figure.figsize': (12,5)})\nplt.figure(figsize = (12,5));","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Option #1 Deleting unused variables and gc.collect()"},{"metadata":{"trusted":true},"cell_type":"code","source":"temp = pd.read_csv('../input/train_sample.csv')\n\ntemp['os'] = temp['os'].astype('str')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#delete when no longer needed\ndel temp\n#collect residual garbage\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Option #2 Presetting the datatypes"},{"metadata":{"trusted":true},"cell_type":"code","source":"dtypes = {'ip':'uint32',\n         'app':'uint16',\n         'device': 'uint16',\n         'os': 'uint16',\n         'channel':'uint16',\n         'is_attributed':'uint8'}\n\ntrain = pd.read_csv('../input/train_sample.csv', dtype = dtypes)\ntrain.info()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Option#3 Importing selected rows of a csv file"},{"metadata":{"trusted":true},"cell_type":"code","source":"train =pd.read_csv('../input/train.csv', nrows = 1000, dtype = dtypes)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#plain skipping looses heading info. Its OK for files that dont have headings\n#or dataframes you'll be linking together, or where you make your own custom headings...\ntrain = pd.read_csv('../input/train.csv',skiprows = 5000000, nrows = 1000000, header = None, dtype = dtypes)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#but if you want to import the headings from the original file\n#skip first 5mil rows, but use the first row for heading:\ntrain = pd.read_csv('../input/train.csv', skiprows = range(1,5000000), nrows = 1000000, dtype = dtypes)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#Genera list of lines to skip\nlines = 5000000\nskiplines = np.random.choice(np.arange(1,lines), size = lines - 1 -1000000, replace = False)\nskiplines = np.sort(skiplines)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"#checking you list\nprint('lines to skip:', len(skiplines))\nprint('remaining lines in sample:', lines - len(skiplines), '(remember that it includes the heading!)')\n\n\n############SANITY CHECK########################\n#find lines that weren't skipped by checking difference berween each conseductive line\n#how many out of first 10000 will be imported into the csv?\ndiff = skiplines[1:100000]-skiplines[2:100001]\nremain = sum(diff!=-1)\nprint('Ratio of lines from first 100000 lines:', '0:.5f'.format(remain/100000))\nprint('Ratio imported from all lines:', '{0:.5f}'.format((lines-len(skiplines))/lines))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv', skiprows = skiplines, dtype = dtypes)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}