{"cells":[{"metadata":{"_uuid":"f61e288d4a6ee2e0cfa2bdce514041917c02216b"},"cell_type":"markdown","source":"I referred to this kernal: \"https://www.kaggle.com/wesamelshamy/high-correlation-feature-image-classification-conf/data\"\nTo solve the problem of disk space, and being not able to analyze image of every images, I tried to make a loop, in which create directory, load image, generate score, and remove the directory. By doing this we could save the disk space but the problem is it takes too long. I think this is inevitable dilemma between storage capacity and processing capacity. I'm trying to use the score (prediction certainty) as one of the features. It seems that the images from the Zip file is not in the order of items in training data csv, so I'm trying to index them using image path. "},{"metadata":{"_uuid":"ea55a0629f55571eea4ae324ce9283c61cdfa011","_cell_guid":"206c03a1-5ddc-4bc9-bb10-d45380768049","trusted":true,"collapsed":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nfrom zipfile import ZipFile\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom dask import bag, threaded\nfrom dask.diagnostics import ProgressBar\nimport matplotlib.pyplot as plt\ndf = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\n\ndf_train = df\ndf_test = test","execution_count":39,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba90df73cde27411a3b13e86f8bda547b6aed708","collapsed":true},"cell_type":"code","source":"# get filenames\nzipped = ZipFile('../input/train_jpg.zip')\nfilenames = zipped.namelist()[1:] # exclude the initial directory listing\nprint(len(filenames))\n\n","execution_count":40,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc8b30f6c49826d57d8df7dc459f64d2bc194e56","collapsed":true},"cell_type":"code","source":"#get blurrness score\n\ndef get_blurrness(file):\n    exfile = zipped.read(file)\n    arr = np.frombuffer(exfile, np.uint8)\n    if arr.size > 0:   # exclude dirs and blanks\n        imz = cv2.imdecode(arr, flags=cv2.COLOR_BGR2GRAY)\n        fm = cv2.Laplacian(imz, cv2.CV_64F).var()\n    else: \n        fm = -1\n    return fm\n\nblurrness = []\niteration = filenames[600000:750000]\nfor i in range(0, len(iteration)):\n    print(i)\n    blurrness.append(get_blurrness(iteration[i]))\n    \nframe = pd.DataFrame({\"File\" : filenames[600000:750000], \"Score\": blurrness})\nframe.to_csv(\"7.csv\", index = False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.5","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}