{"cells":[{"metadata":{"_cell_guid":"9303cdcb-ede1-4cee-a4ea-ddb81bcb8d14","_uuid":"a35c3bdde841229f9358fe3a31973fc0254e82da"},"cell_type":"markdown","source":"Submision file for this competition is about 500M, so it makes sense to minimize it. The traditional way is to use \"float_format\" parameter of pandas \"to_csv\" method + compression. \nThe problem with \"float_format\" is that different predictions may become equal ones - this situation depends on specific model and it is hard to predict what can be impact to the score (may assume it is minimal, but who knows). \n\nThis kernel shows how submission file can be minimized, while ensuring that the score is not changed.\nIt is based on the following ideas:\n\n1. ROC AUC stays the same if relative position of predictions remain the same. In other words, if one sorts predictions by is_attributed and change predictions so their position in sorted list remains, then AUC is not changed.\n2. Usually number of unique predictions are less than total number of rows in predictions (in my case unique predictions is about half  of rows)\n3. Number of digits after decimal point can be decreased. Extra precision is not needed here, minimization procedure should guarantee only that different predictions remain different.\n4. Leading and trailing zeros can be omitted\n\nPotential disadvantage: after such minimization, blending may (or may not) get different result. So keep original submission.\n\nThe kernel uses \"FTRL revisited 22\" data as an example. Plain and compressed size for original, formatted (float_format) and minimized (this algorithm) submission files are calculated."},{"metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","collapsed":true,"trusted":false},"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport math\n\n\ndef get_minimized(submission):\n    \"\"\"\n    Minimizes size of column 'is_attributed' from submission\n    :param submission: panda DataFrame\n    :return: minimized column 'is_attributed' as pandas Series\n    \"\"\"\n    unique_values = np.sort(submission.is_attributed.unique())\n    size = unique_values.shape[0]\n    digits = int(math.ceil(math.log10(size)))\n    print('Unique size {:,}, digits: {}'.format(size, digits))\n    step = 10 ** -digits\n    format_string = '{:.' + str(digits) + 'f}'\n    mapping = {}\n    value = step\n    for i in range(size):\n        original = unique_values[i]\n        text_value = format_string.format(value).strip('0')\n        mapping[original] = text_value\n        value += step\n\n    minimized = submission['is_attributed'].map(mapping.get)\n    return minimized\n\n\ndef save_submission(file_path, submission, compression=None, line_terminator='\\r'):\n    submission.to_csv(file_path, index=False, line_terminator=line_terminator, chunksize=1024, compression=compression)","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","collapsed":true,"trusted":false},"cell_type":"code","source":"sample_file_path = '../input/ftrl-revisited-22/sub_proba.csv'\nminimized_file_name = 'minimized.csv'\nformatted_file_name = 'formatted.csv'\n\n\nsubmission = pd.read_csv(sample_file_path, dtype={'click_id': 'int32', 'is_attributed': 'float64'}, engine='c',\n                         na_filter=False, memory_map=True)\n\n# save file with formatting\nsubmission.to_csv(formatted_file_name, float_format='%.8f', index=False)\nformatted_size = os.path.getsize(formatted_file_name)\nos.remove(formatted_file_name)\n\n# save file with formatting with compression\nformatted_file_name_gz = formatted_file_name + \".bz2\"\nsubmission.to_csv(formatted_file_name_gz, float_format='%.8f', index=False, compression='bz2')\nformatted_size_gz = os.path.getsize(formatted_file_name_gz)\nos.remove(formatted_file_name_gz)\n\nminimized = get_minimized(submission)\n\nsubmission['is_attributed'] = minimized\n\n# save minimized file\nsave_submission(minimized_file_name, submission)\nminimized_size = os.path.getsize(minimized_file_name)\nos.remove(minimized_file_name)\n\n# save minimized with compression\nminimized_file_name_gz = minimized_file_name + '.bz2'\nsave_submission(minimized_file_name_gz, submission, compression='bz2')\nminimized_size_gz = os.path.getsize(minimized_file_name_gz)\nos.remove(minimized_file_name_gz)\n\noriginal_size = os.path.getsize(sample_file_path)\n\nprint('original file size: {:,}'.format(original_size))\nprint('minimized file size: {:,}'.format(minimized_size))\nprint('formatted file size: {:,}'.format(formatted_size))\nprint('minimized + compressed file size: {:,}'.format(minimized_size_gz))\nprint('formatted + compressed file size: {:,}'.format(formatted_size_gz))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}