{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.9","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":13032,"databundleVersionId":862545,"sourceType":"competition"}],"dockerImageVersionId":30055,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/input/imaterialist-fashion-2019-FGVC6/label_descriptions.json\",\"r\") as file:\n    data = json.load(file)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"../input/imaterialist-fashion-2019-FGVC6/train\"\ntrain_csv = \"../input/imaterialist-fashion-2019-FGVC6/train.csv\"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(train_csv)\ndf.head(15)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.ClassId.value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking if there are attributes are 91\n\n# test_dataframe = df[\"ClassId\"].values\n# for x in test_dataframe:\n# #     if \"91\" in x\n#         print(\"91\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom random import randint,choice","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataframe = df[\"ClassId\"].values\ntest_dataframe","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nimport os","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def function_to_print_attributes_by_class_id(dataframe,train_dir,class_id,name,num_examples=2):\n#     byClass_Dataframe = dataframe[dataframe[\"ClassId\"]==class_id]\n#     to_plt = byClass_Dataframe.head(randint(1,10))[\"ImageId\"].values[0]\n#     img = plt.imread(os.path.join(train_dir,to_plt))\n#     plt.imshow(img)\n    test_dataframe = dataframe[\"ClassId\"].values\n    for i in range(num_examples):\n        random_num = choice([ i for (i,data) in enumerate(test_dataframe) if class_id in data])\n        to_plt = dataframe[random_num:random_num+1][\"ImageId\"].values[0]\n        img = cv2.imread(os.path.join(train_dir,to_plt))\n        cv2.imwrite(\"./\"+name +str(i)+\".jpg\",img)\n        plt.imshow(cv2.cvtColor(img,cv2.COLOR_BGR2RGB))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attrib = data[\"attributes\"]\nfor i,key in enumerate(attrib):\n    if key[\"id\"]==i:\n        string = key[\"name\"]\n        function_to_print_attributes_by_class_id(df,train_dir,str(i),string,num_examples=2)\n#         print(i,string)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"attributes = {}\nfor x in data[\"attributes\"]:\n    if x[\"supercategory\"] not in attributes:\n        attributes[x[\"supercategory\"]] = []\n        attributes[x[\"supercategory\"]].append(x[\"name\"])\n    else:\n        attributes[x[\"supercategory\"]].append(x[\"name\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(attributes)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categories = {}\nfor x in data[\"categories\"]:\n    if x[\"supercategory\"] not in categories:\n        categories[x[\"supercategory\"]] = []\n        categories[x[\"supercategory\"]].append(x[\"name\"])\n    else:\n        categories[x[\"supercategory\"]].append(x[\"name\"])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(categories)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_id = df[\"ClassId\"].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_id.head(30)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_by_class_id(dataframe,train_dir,class_id):\n    byClass_Dataframe = dataframe[dataframe[\"ClassId\"]==class_id]\n    rand = randint(0,byClass_Dataframe.shape[0])\n    to_plt = byClass_Dataframe[rand:rand+1][\"ImageId\"].values[0]\n#     to_plt = choice(byClass_Dataframe[0])\n#     print(to_plt)\n    img = plt.imread(os.path.join(train_dir,to_plt))\n    plt.imshow(img)\n\nprint_by_class_id(df,train_dir,\"32\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count(dataframe,class_id):\n#     byClass_Dataframe = dataframe[dataframe[\"ClassId\"]==class_id]\n#     to_plt = byClass_Dataframe.head(randint(1,10))[\"ImageId\"].values[0]\n#     img = plt.imread(os.path.join(train_dir,to_plt))\n#     plt.imshow(img)\n    test_dataframe = dataframe[\"ClassId\"].values\n    total_data = [ i for (i,data) in enumerate(test_dataframe) if class_id in data]\n#     random_num = choice(total_data)\n#     to_plt = dataframe[random_num:random_num+1][\"ImageId\"].values[0]\n    return len(total_data)\ncount(df,\"0\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lists = []\nnames = []\nfor i in data[\"attributes\"]:\n    ids = i['id']\n    lists.append(count(df,str(ids)))\n    \n    name = i[\"name\"]\n    names.append(name)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"counting_dict = {}\nfor a,b in zip(names,lists):\n    counting_dict[a] = b","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for sub_categories in attributes:\n    count = 0\n    for sub_sub in attributes[sub_categories]:\n        count+=counting_dict[sub_sub]\n    print(count,sub_categories)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nfrom skimage import io\nfrom skimage.filters import gaussian\nimport zipfile\nfrom io import BytesIO\nimport os\n\n# Function to downsample and blur the image\ndef downsample_and_blur(img, skip, sigma=1):\n    downscaled_img = img[::skip, ::skip]\n    blurred_img = gaussian(downscaled_img, sigma=sigma, preserve_range=True)\n    return blurred_img.astype(np.uint8)\n\n# Directory where test images are located\ninput_directory = \"/kaggle/input/imaterialist-fashion-2019-FGVC6/test/\"\n\n# List all files in the directory\nfiles = os.listdir(input_directory)\n\n# Process images and save into separate zip files for original and processed images\noriginal_zip_buffer = BytesIO()\nprocessed_zip_buffer = BytesIO()\n\nwith zipfile.ZipFile(original_zip_buffer, 'a', compression=zipfile.ZIP_DEFLATED) as original_zip_file, \\\n        zipfile.ZipFile(processed_zip_buffer, 'a', compression=zipfile.ZIP_DEFLATED) as processed_zip_file:\n    \n    skip = 7\n    sigma = 0.5\n    \n    for img_index, filename in enumerate(files):\n        # Read image\n        img_path = os.path.join(input_directory, filename)\n        img = io.imread(img_path)\n        \n        # Save original image\n        original_filename = f\"img{img_index + 1}.jpg\"\n        io.imsave(original_filename, img)\n        original_zip_file.write(original_filename)\n        \n        # Process image\n        blurred_img = downsample_and_blur(img, skip, sigma)\n        \n        # Save processed image\n        processed_filename = f\"img{img_index + 1}_processed.jpg\"\n        io.imsave(processed_filename, blurred_img)\n        processed_zip_file.write(processed_filename)\n\n# Save the zip files\noriginal_output_zip_filename = '/kaggle/working/original_images.zip'\nprocessed_output_zip_filename = '/kaggle/working/processed_images.zip'\n\nwith open(original_output_zip_filename, 'wb') as original_f, open(processed_output_zip_filename, 'wb') as processed_f:\n    original_f.write(original_zip_buffer.getvalue())\n    processed_f.write(processed_zip_buffer.getvalue())\n\n# Print message indicating successful creation of zip files\nprint(\"Zip files created successfully.\")\n\n# Note: Downloading zip files in Kaggle may require some additional steps.\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T20:30:43.849591Z","iopub.execute_input":"2024-04-24T20:30:43.850144Z","iopub.status.idle":"2024-04-24T20:57:40.294515Z","shell.execute_reply.started":"2024-04-24T20:30:43.849976Z","shell.execute_reply":"2024-04-24T20:57:40.293497Z"},"trusted":true},"execution_count":null,"outputs":[]}]}