{"metadata": {"language_info": {"codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3", "mimetype": "text/x-python", "nbconvert_exporter": "python", "name": "python", "file_extension": ".py", "version": "3.6.1"}, "kernelspec": {"display_name": "Python 3", "name": "python3", "language": "python"}}, "nbformat": 4, "nbformat_minor": 0, "cells": [{"metadata": {"_uuid": "9c97ebd32f97b46625a56068bc2eb0355b9f3e84", "collapsed": false, "_cell_guid": "9e18f4fc-c3f7-48d4-b4cd-47246ea9ffbe", "_execution_state": "idle"}, "source": "These are slightly better versions of the (unofficial) functions to read and view the competition files. Most of the information in the file header is identical across scans and thus not relevant to the competition.\n\n## Changes ##\nread_header is made faster using struct unpack.\n\nread_header only reads 512 bytes (vs orig 532) spare_end is changed from float32 to int16 as previous spares were int16 I infered this to be a mistake.\n\nread_header no longer returns arrays of scalars; makes life better for pandas.\n\nread_data is refactored removing redundant code", "outputs": [], "execution_count": null, "cell_type": "markdown"}, {"metadata": {"_uuid": "ff946ee0551084d70e00f3c2d807da538472462b", "trusted": false, "_cell_guid": "d554894d-f6fb-49f3-9ede-5193242007c0", "_execution_state": "idle"}, "source": "import os\nimport numpy as np\nimport pandas as pd\nfrom struct import unpack", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "1275b2e6fa0aa36fbec71c9ab0fe319958c55593", "collapsed": false, "trusted": false, "_cell_guid": "f52e7f9f-c7e4-4606-bf77-e64443a3f5a7", "_execution_state": "idle"}, "source": "#PATH = '/data/passenger-screening/data/interim/'\n#FILES = [file for file in os.listdir(PATH) if file != '.gitkeep']", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "0b19cc11a3daca36e161b087177ca8fbd5a1ede7", "collapsed": false, "_cell_guid": "365182b0-d584-4574-a1a6-46ef068df0e3", "_execution_state": "idle"}, "source": "## Read header ##", "outputs": [], "execution_count": null, "cell_type": "markdown"}, {"metadata": {"_uuid": "c28a1c2a6aa8c89ba8ce01b228c80cedada7d4ba", "collapsed": false, "trusted": false, "_cell_guid": "4ca3d23b-7b9c-439a-99be-03396f54ea0b", "_execution_state": "idle"}, "source": "def read_header(file, path): # good default value path=PATH\n    \"\"\"Read image header (first 512 bytes)\"\"\"\n    h = dict()\n    with open(path+file, 'rb') as f:\n        header = f.read(512)\n    h['filename'], h['parent_filename'], h['comments1'], h['comments2'] = unpack('20s20s80s80s', header[:200])\n    h['energy_type'], h['config_type'], h['file_type'], h['trans_type'], h['scan_type'], h['data_type'] = unpack('6h', header[200:212])\n    h['date_modified'], h['frequency'], h['mat_velocity'], h['num_pts'] = unpack('16s2fi', header[212:240])\n    h['num_polarization_channels'], h['spare00'] = unpack('2h', header[240:244])\n    h['adc_min_voltage'], h['adc_max_voltage'], h['band_width'] = unpack('3f', header[244:256])\n    h['spare01'], h['spare02'], h['spare03'], h['spare04'], h['spare05'] = unpack('5h', header[256:266])\n    h['polar_t1'], h['polar_t2'], h['polar_t3'], h['polar_t4'] = unpack('4h', header[266:274])\n    h['record_header_size'], h['word_type'], h['word_precision'] = unpack('3h', header[274:280])\n    h['min_data_value'], h['max_data_value'], h['avg_data_value'], h['data_scale_factor'] = unpack('4f', header[280:296])\n    h['data_units'] = unpack('h', header[296:298])\n    h['surf_removal'], h['edge_weighting'], h['x_units'], h['y_units'], h['z_units'], h['t_units'] = unpack('6H', header[298:310])\n    h['spare06'] = unpack('h', header[310:312])\n    h['x_return_speed'], h['y_return_speed'], h['z_return_speed'] = unpack('3f', header[312:324])\n    h['scan_orientation'], h['scan_direction'], h['data_storage_order'], h['scanner_type'] = unpack('4h', header[324:332])\n    h['x_inc'], h['y_inc'], h['z_inc'], h['t_inc'] = unpack('4f', header[332:348])\n    h['num_x_pts'], h['num_y_pts'], h['num_z_pts'], h['num_t_pts'] = unpack('4i', header[348:364])\n    h['x_speed'], h['y_speed'], h['z_speed'] = unpack('3f', header[364:376])\n    h['x_acc'], h['y_acc'], h['z_acc'] = unpack('3f', header[376:388])\n    h['x_motor_res'], h['y_motor_res'], h['z_motor_res'] = unpack('3f', header[388:400])\n    h['x_encoder_res'], h['y_encoder_res'], h['z_encoder_res'] = unpack('3f', header[400:412])\n    h['date_processed'], h['time_processed'] = unpack('8s8s', header[412:428])\n    h['depth_recon'], h['x_max_travel'], h['y_max_travel'], h['elevation_offset_angle'] = unpack('4f', header[428:444])\n    h['roll_offset_angle'], h['z_max_travel'], h['azimuth_offset_angle'] = unpack('3f', header[444:456])\n    h['adc_type'], h['spare06'], h['scanner_radius'] = unpack('2hf', header[456:464])\n    h['x_offset'], h['y_offset'], h['z_offset'], h['t_delay'] = unpack('4f', header[464:480])\n    h['range_gate_start'], h['range_gate_end'], h['ahis_software_version'] = unpack('3f', header[480:492])\n    h['spare07'], h['spare08'], h['spare09'], h['spare10'], h['spare11'], h['spare12'], \\\n    h['spare13'], h['spare14'], h['spare15'], h['spare16'] = unpack('10h', header[492:])\n    return h", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "97dad5c9be4ff1c791efa9fcf3cc98788784e898", "collapsed": false, "trusted": false, "_cell_guid": "0783968c-a237-4543-8a6f-ec2aada38cf3", "_execution_state": "idle"}, "source": "def read_header_orig(file, path): # good default value path=PATH\n    \"\"\"Read image header (first 512 bytes)\"\"\"\n    h = dict()\n    with open(path+file, 'rb') as fid:\n        h['filename'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 20))\n        h['parent_filename'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 20))\n        h['comments1'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 80))\n        h['comments2'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 80))\n        h['energy_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['config_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['file_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['trans_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['scan_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['data_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['date_modified'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 16))\n        h['frequency'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['mat_velocity'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['num_pts'] = np.fromfile(fid, dtype = np.int32, count = 1)\n        h['num_polarization_channels'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['spare00'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['adc_min_voltage'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['adc_max_voltage'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['band_width'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['spare01'], h['spare02'], h['spare03'], h['spare04'], h['spare05'] = np.fromfile(fid, dtype = np.int16, count = 5)\n        h['polar_t1'], h['polar_t2'], h['polar_t3'], h['polar_t4'] = np.fromfile(fid, dtype = np.int16, count = 4)\n        h['record_header_size'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['word_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['word_precision'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['min_data_value'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['max_data_value'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['avg_data_value'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['data_scale_factor'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['data_units'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['surf_removal'] = np.fromfile(fid, dtype = np.uint16, count = 1)\n        h['edge_weighting'] = np.fromfile(fid, dtype = np.uint16, count = 1)\n        h['x_units'] = np.fromfile(fid, dtype = np.uint16, count = 1)\n        h['y_units'] = np.fromfile(fid, dtype = np.uint16, count = 1)\n        h['z_units'] = np.fromfile(fid, dtype = np.uint16, count = 1)\n        h['t_units'] = np.fromfile(fid, dtype = np.uint16, count = 1)\n        h['spare06'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['x_return_speed'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_return_speed'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_return_speed'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['scan_orientation'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['scan_direction'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['data_storage_order'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['scanner_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['x_inc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_inc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_inc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['t_inc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['num_x_pts'] = np.fromfile(fid, dtype = np.int32, count = 1)\n        h['num_y_pts'] = np.fromfile(fid, dtype = np.int32, count = 1)\n        h['num_z_pts'] = np.fromfile(fid, dtype = np.int32, count = 1)\n        h['num_t_pts'] = np.fromfile(fid, dtype = np.int32, count = 1)\n        h['x_speed'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_speed'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_speed'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['x_acc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_acc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_acc'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['x_motor_res'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_motor_res'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_motor_res'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['x_encoder_res'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_encoder_res'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_encoder_res'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['date_processed'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 8))\n        h['time_processed'] = b''.join(np.fromfile(fid, dtype = 'S1', count = 8))\n        h['depth_recon'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['x_max_travel'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_max_travel'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['elevation_offset_angle'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['roll_offset_angle'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_max_travel'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['azimuth_offset_angle'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['adc_type'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['spare06'] = np.fromfile(fid, dtype = np.int16, count = 1)\n        h['scanner_radius'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['x_offset'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['y_offset'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['z_offset'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['t_delay'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['range_gate_start'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['range_gate_end'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['ahis_software_version'] = np.fromfile(fid, dtype = np.float32, count = 1)\n        h['spare07'], h['spare08'], h['spare09'], h['spare10'], h['spare11'], h['spare12'], \\\n        h['spare13'], h['spare14'], h['spare15'], h['spare16'] = np.fromfile(fid, dtype = np.int16, count = 10)\n    return h", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "8faaf1f65aa7870da131e8cb1c9bf46c7915dc38", "collapsed": false, "trusted": false, "_cell_guid": "7181e987-d726-4a23-9cbe-6006019adda5", "_execution_state": "idle"}, "source": "#%timeit [read_header(file) for file in FILES]\n#%timeit pd.DataFrame([read_header_orig(file) for file in FILES])\n#%timeit pd.DataFrame([read_header(file) for file in FILES])", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "70c9e12c4c2444ef51220d64476386d4e307ef56", "collapsed": false, "trusted": false, "_cell_guid": "b49924e2-83e2-4506-8b7b-690cb55c028a", "_execution_state": "idle"}, "source": "#df = pd.DataFrame([read_header_orig(file) for file in FILES])\n#df.head()", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "be21904165a98a126b0434135b39da51879424af", "collapsed": false, "trusted": false, "_cell_guid": "a1c69272-ca05-4e5f-843d-ad158e4a8768", "_execution_state": "idle"}, "source": "#df = pd.DataFrame([read_header(file) for file in FILES])\n#df.head()", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "6ca37056b03966d3378adbcb733ccbf6e6a39d50", "collapsed": false, "_cell_guid": "a3a46ced-d5a5-4229-836b-1ad1c8731103", "_execution_state": "idle"}, "source": "## Read image ##", "outputs": [], "execution_count": null, "cell_type": "markdown"}, {"metadata": {"_uuid": "d7df77b4128076a10381fe18991605e4436fc3b5", "collapsed": false, "trusted": false, "_cell_guid": "64e57e33-8e54-43cb-9226-8ece9b076a6d", "_execution_state": "busy"}, "source": "def read_data(file, path): # good default value path=PATH\n    \"\"\"Read any of the 4 types of image files, returns a numpy array of the image contents\"\"\"\n    extension = file.split('.')[-1]\n    with open(path+file, 'rb') as f:\n        header = f.read(512)\n        word_type = unpack('h', header[276:278])\n        scale_factor = unpack('f', header[292:296])\n        nx, ny, nz, nt = unpack('4i', header[348:364])\n        if extension == 'ahi':\n            data = np.fromfile(f, dtype=np.float32, count=2*nx*ny*nt)\n            data = data.reshape(2, ny, nx, nt, order='F')\n            return data[0,:,:,:], data[1,:,:,:]\n        else:\n            if word_type == 7: #float32\n                data = np.fromfile(f, dtype=np.float32, count=nx*ny*nt)\n            else:\n                data = np.fromfile(f, dtype=np.uint16, count=nx*ny*nt)\n            data = data * scale_factor\n            if extension == 'a3d':\n                return data.reshape(nx, nt, ny, order='F')\n            return data.reshape(nx, ny, nt, order='F')", "outputs": [], "execution_count": null, "cell_type": "code"}, {"metadata": {"_uuid": "e4725bd5ac4462f6c4f49792f610ea2c17de4c43", "collapsed": false, "trusted": false, "_cell_guid": "362a0975-34fd-4fde-b580-eba3601e2c3b", "_execution_state": "idle"}, "source": "#img = read_data(FILES[0])\n#img.shape", "outputs": [], "execution_count": null, "cell_type": "code"}]}