{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport pickle\nimport numpy as np\nimport pandas as pd","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.393157,"end_time":"2022-03-21T06:27:31.224087","exception":false,"start_time":"2022-03-21T06:27:30.83093","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-22T08:25:32.509037Z","iopub.execute_input":"2022-03-22T08:25:32.50988Z","iopub.status.idle":"2022-03-22T08:25:32.952377Z","shell.execute_reply.started":"2022-03-22T08:25:32.509761Z","shell.execute_reply":"2022-03-22T08:25:32.951428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def breaker(num: int = 50, char: str = \"*\") -> None:\n    print(\"\\n\" + num*char + \"\\n\")\n\n\ndef preprocess(image: np.ndarray) -> np.ndarray:\n    return cv2.cvtColor(src=image, code=cv2.COLOR_BGR2RGB)\n\n\ndef get_statistics(path: str, names: np.ndarray, sizes: list, broken_images: list) -> list:\n    r_means, g_means, b_means, r_stds, g_stds, b_stds = [], [], [], [], [], []\n\n    i = 0\n    for size in sizes:\n        r_mean, g_mean, b_mean, r_std, g_std, b_std = 0.0, 0.0, 0.0, 0.0, 0.0, 0.0\n        for name in names:\n            if name in broken_images:\n                pass\n            else:\n                main_image = preprocess(cv2.imread(os.path.join(path, name), cv2.IMREAD_COLOR))\n                image = cv2.resize(src=main_image, dsize=(size, size), interpolation=cv2.INTER_AREA)\n                r_mean += image[:, :, 0].mean()\n                g_mean += image[:, :, 1].mean()\n                b_mean += image[:, :, 2].mean()\n                r_std  += image[:, :, 0].std()\n                g_std  += image[:, :, 1].std()\n                b_std  += image[:, :, 2].std()\n        r_means.append(r_mean / len(names))\n        g_means.append(g_mean / len(names))\n        b_means.append(b_mean / len(names))\n\n        r_stds.append(r_std / len(names))\n        g_stds.append(g_std / len(names))\n        b_stds.append(b_std / len(names))\n    \n    return r_means, g_means, b_means, r_stds, g_stds, b_stds ","metadata":{"papermill":{"duration":0.023802,"end_time":"2022-03-21T06:27:31.255123","exception":false,"start_time":"2022-03-21T06:27:31.231321","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-22T08:28:42.97062Z","iopub.execute_input":"2022-03-22T08:28:42.971442Z","iopub.status.idle":"2022-03-22T08:28:42.981828Z","shell.execute_reply.started":"2022-03-22T08:28:42.971399Z","shell.execute_reply":"2022-03-22T08:28:42.98115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/sorghum-id-fgvc-9/train_cultivar_mapping.csv\")\n\nbroken_images = [filename for filename in train_df.image if filename not in os.listdir(\"../input/sorghum-id-fgvc-9/train_images\")]\nfor broken_image in broken_images:\n    index = train_df.index[train_df.image == broken_image]\n    train_df = train_df.drop(index=index)\n    \nfilenames = train_df.iloc[:, 0].copy().values","metadata":{"papermill":{"duration":0.121298,"end_time":"2022-03-21T06:27:31.382822","exception":false,"start_time":"2022-03-21T06:27:31.261524","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-22T08:28:44.7506Z","iopub.execute_input":"2022-03-22T08:28:44.750923Z","iopub.status.idle":"2022-03-22T08:28:44.775984Z","shell.execute_reply.started":"2022-03-22T08:28:44.75089Z","shell.execute_reply":"2022-03-22T08:28:44.775047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sizes = [256, 384, 512, 768, 1024]\n\nr_means, g_means, b_means, r_stds, g_stds, b_stds = get_statistics(\"../input/sorghum-id-fgvc-9/train_images\", filenames, sizes, broken_images)","metadata":{"papermill":{"duration":4953.948455,"end_time":"2022-03-21T07:50:05.337827","exception":false,"start_time":"2022-03-21T06:27:31.389372","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-03-22T08:28:46.856129Z","iopub.execute_input":"2022-03-22T08:28:46.856574Z","iopub.status.idle":"2022-03-22T08:28:46.88108Z","shell.execute_reply.started":"2022-03-22T08:28:46.856531Z","shell.execute_reply":"2022-03-22T08:28:46.879829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"breaker()\nfor i in range(len(sizes)):\n    print(f\"Red Channel Mean ({sizes[i]})   : {r_means[i]:.5f}\")\n    print(f\"Green Channel Mean ({sizes[i]}) : {g_means[i]:.5f}\")\n    print(f\"Blue Channel Mean ({sizes[i]})  : {b_means[i]:.5f}\")\n    print(\"\\n\")\n    print(f\"Red Channel Std ({sizes[i]})    : {r_stds[i]:.5f}\")\n    print(f\"Green Channel Std ({sizes[i]})  : {g_stds[i]:.5f}\")\n    print(f\"Blue Channel Std ({sizes[i]})   : {b_stds[i]:.5f}\")\n    breaker()","metadata":{"execution":{"iopub.execute_input":"2022-03-21T07:50:05.358233Z","iopub.status.busy":"2022-03-21T07:50:05.3576Z","iopub.status.idle":"2022-03-21T07:50:05.378871Z","shell.execute_reply":"2022-03-21T07:50:05.378291Z"},"papermill":{"duration":0.031272,"end_time":"2022-03-21T07:50:05.379006","exception":false,"start_time":"2022-03-21T07:50:05.347734","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}