{"nbformat_minor": 1, "metadata": {"kernelspec": {"name": "python3", "display_name": "Python 3", "language": "python"}, "language_info": {"nbconvert_exporter": "python", "codemirror_mode": {"name": "ipython", "version": 3}, "version": "3.6.1", "mimetype": "text/x-python", "file_extension": ".py", "pygments_lexer": "ipython3", "name": "python"}}, "nbformat": 4, "cells": [{"source": ["## Check hashsum of files\n", "\n", "Download doesn't end at all, so this was made."], "metadata": {"_cell_guid": "cdb70995-66f6-4f62-acf4-5c00ee251f9c", "_uuid": "3d97f81ebfbbb88526450c8b6bc47443832d5c74"}, "cell_type": "markdown"}, {"execution_count": null, "metadata": {"_cell_guid": "7bff90df-5989-4074-aa9c-14f6a5c477b3", "_uuid": "4fc01e20ee16c077ef6702d8942b855c5750f764", "collapsed": true}, "source": ["import hashlib\n", "import pandas as pd\n", "\n", "\n", "file_names = open('../input/category_names.csv', 'rb').read()\n", "md5_names  = hashlib.md5(file_names).hexdigest()\n", "sha1_names = hashlib.sha1(file_names).hexdigest()\n", "del file_names\n", "\n", "file_subex = open('../input/sample_submission.csv', 'rb').read()\n", "md5_subex  = hashlib.md5(file_subex).hexdigest()\n", "sha1_subex = hashlib.sha1(file_subex).hexdigest()\n", "del file_subex\n", "\n", "res_cmd    = !md5sum ../input/test.bson\n", "md5_test   = str(res_cmd).split('\\'')[1].split(' ')[0]\n", "res_cmd    = !sha1sum ../input/test.bson\n", "sha1_test  = str(res_cmd).split('\\'')[1].split(' ')[0]\n", "\n", "res_cmd    = !md5sum ../input/train.bson\n", "md5_train  = str(res_cmd).split('\\'')[1].split(' ')[0]\n", "res_cmd    = !sha1sum ../input/train.bson\n", "sha1_train = str(res_cmd).split('\\'')[1].split(' ')[0]\n", "\n", "file_ex    = open('../input/train_example.bson', 'rb').read()\n", "md5_ex     = hashlib.md5(file_ex).hexdigest()\n", "sha1_ex    = hashlib.sha1(file_ex).hexdigest()\n", "del file_ex\n", "\n", "df = pd.DataFrame([\n", "    pd.Series([md5_names, sha1_names], index=['MD5', 'SHA-1']),\n", "    pd.Series([md5_subex, sha1_subex], index=['MD5', 'SHA-1']),\n", "    pd.Series([md5_test,  sha1_test],  index=['MD5', 'SHA-1']),\n", "    pd.Series([md5_train, sha1_train], index=['MD5', 'SHA-1']),\n", "    pd.Series([md5_ex,    sha1_ex],    index=['MD5', 'SHA-1']),\n", "])\n", "\n", "df.index = ['category_names.csv', 'sample_submission.csv', 'test.bson', 'train.bson', 'train_example.bson']\n", "df.to_csv('hashsum_cdiscount.csv')\n", "df"], "cell_type": "code", "outputs": []}]}