{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n%matplotlib inline\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport cv2\nimport os\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import OneHotEncoder\nimport time\n\nstart = time.time()\n\ntrain_image_path = '../input/resized-plant2021/img_sz_256/' #uses smaller version of dataset for efficiency\ntest_image_path = '../input/plant-pathology-2021-fgvc8/test_images/'\ntrain_df_path = '../input/plant-pathology-2021-fgvc8/train.csv'\ntest_df_path = '../input/plant-pathology-2021-fgvc8/sample_submission.csv'\n\ntrain_df = pd.read_csv(train_df_path)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-12-03T10:20:19.548150Z","iopub.execute_input":"2021-12-03T10:20:19.548475Z","iopub.status.idle":"2021-12-03T10:20:21.119803Z","shell.execute_reply.started":"2021-12-03T10:20:19.548390Z","shell.execute_reply":"2021-12-03T10:20:21.119054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"blue_minimums = []\nblue_maximums = []\nblue_means = []\n\ngreen_minimums = []\ngreen_maximums = []\ngreen_means = []\n\nred_minimums = []\nred_maximums = []\nred_means = []\n\n\nfor image_path in train_df['image'].tolist():\n    img = cv2.imread(train_image_path + image_path) #cv2 reads image into numpy array\n    #openCV uses BGR image formatting, so\n    blue_channel = img[:,:,0]\n    green_channel = img[:,:,1]\n    red_channel = img[:,:,2]\n    \n\n    \n    #extract features\n    blue_minimums.append(np.min(blue_channel))\n    blue_maximums.append(np.max(blue_channel).astype(np.int16))\n    blue_means.append(np.mean(blue_channel))\n    \n    green_minimums.append(np.min(green_channel))\n    green_maximums.append(np.max(green_channel).astype(np.int16))\n    green_means.append(np.mean(green_channel))\n\n    red_minimums.append(np.min(red_channel))\n    red_maximums.append(np.max(red_channel).astype(np.int16))\n    red_means.append(np.mean(red_channel))","metadata":{"execution":{"iopub.status.busy":"2021-12-03T10:20:21.121329Z","iopub.execute_input":"2021-12-03T10:20:21.121534Z","iopub.status.idle":"2021-12-03T10:23:26.568929Z","shell.execute_reply.started":"2021-12-03T10:20:21.121507Z","shell.execute_reply":"2021-12-03T10:23:26.567711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colour_features = pd.DataFrame(\n    {     \n    'blue_minimums' : blue_minimums,\n    'blue_maximums' : blue_maximums,\n    'blue_means' : blue_means,\n    \n    'green_minimums' : green_minimums,\n    'green_maximums' : green_maximums,\n    'green_means' : green_means,\n\n    'red_minimums' : red_minimums,\n    'red_maximums' : red_maximums,\n    'red_means' : red_means\n    })\n\ny_train = train_df['labels'].astype('category')\n\nlinreg = LogisticRegression(solver='liblinear')\nlinreg.fit(colour_features, y_train)","metadata":{"execution":{"iopub.status.busy":"2021-12-03T10:23:26.570071Z","iopub.execute_input":"2021-12-03T10:23:26.570283Z","iopub.status.idle":"2021-12-03T10:23:27.643667Z","shell.execute_reply.started":"2021-12-03T10:23:26.570257Z","shell.execute_reply":"2021-12-03T10:23:27.642815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(colour_features)\nweight = linreg.coef_ \nprint(weight)","metadata":{"execution":{"iopub.status.busy":"2021-12-03T10:23:27.645516Z","iopub.execute_input":"2021-12-03T10:23:27.645740Z","iopub.status.idle":"2021-12-03T10:23:27.667142Z","shell.execute_reply.started":"2021-12-03T10:23:27.645713Z","shell.execute_reply":"2021-12-03T10:23:27.665420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv(test_df_path)\n\nblue_minimums = []\nblue_maximums = []\nblue_means = []\n\ngreen_minimums = []\ngreen_maximums = []\ngreen_means = []\n\nred_minimums = []\nred_maximums = []\nred_means = []\n\nfor image_path in test_df['image'].tolist():\n    img = cv2.imread(test_image_path + image_path) #cv2 reads image into numpy array\n    #openCV uses BGR image formatting, so\n    blue_channel = img[:,:,0]\n    green_channel = img[:,:,1]\n    red_channel = img[:,:,2]\n    \n    #extract features\n    blue_minimums.append(np.min(blue_channel))\n    blue_maximums.append(np.max(blue_channel).astype(np.int16))\n    blue_means.append(np.mean(blue_channel))\n    \n    green_minimums.append(np.min(green_channel))\n    green_maximums.append(np.max(green_channel).astype(np.int16))\n    green_means.append(np.mean(green_channel))\n\n    red_minimums.append(np.min(red_channel))\n    red_maximums.append(np.max(red_channel).astype(np.int16))\n    red_means.append(np.mean(red_channel))\n    \ncolour_features = pd.DataFrame(\n        {     \n        'blue_minimums' : blue_minimums,\n        'blue_maximums' : blue_maximums,\n        'blue_means' : blue_means,\n\n        'green_minimums' : green_minimums,\n        'green_maximums' : green_maximums,\n        'green_means' : green_means,\n\n        'red_minimums' : red_minimums,\n        'red_maximums' : red_maximums,\n        'red_means' : red_means\n        })    \n\npreds = linreg.predict(colour_features)\ntest_df['labels'] = preds\n    \ntest_df.to_csv('submission.csv', index=False)\n\nend = time.time()\nprint(end - start)","metadata":{"execution":{"iopub.status.busy":"2021-12-03T10:23:27.668545Z","iopub.execute_input":"2021-12-03T10:23:27.668937Z","iopub.status.idle":"2021-12-03T10:23:28.692697Z","shell.execute_reply.started":"2021-12-03T10:23:27.668895Z","shell.execute_reply":"2021-12-03T10:23:28.692027Z"},"trusted":true},"execution_count":null,"outputs":[]}]}