{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\nimport numpy as np\nimport pandas as pd\nimport gc\nimport os\nimport PIL\nimport glob\nimport cv2\nimport time\n\nfrom scipy import stats\nfrom multiprocessing import Pool\nfrom PIL import ImageOps,ImageFilter,Image\nfrom tqdm import tqdm\nfrom wordcloud import WordCloud\n\ntqdm.pandas()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_size_df = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"width = []\nheight = []\nfor name in tqdm(train_size_df['id']):\n    img = Image.open('../input/train/'+name+'.png')\n    width.append(img.size[0])\n    height.append(img.size[1])\ntrain_size_df['width'] = width\ntrain_size_df['height'] = height","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_size_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"width_height_ratio = np.zeros(len(train_size_df))\nheight_width_ratio = np.zeros(len(train_size_df))\nfor row_index,(w,h) in tqdm(enumerate(zip(train_size_df['width'],train_size_df['height']))):\n    if w==300:\n        times = h/w\n        times = round(times,1)\n        height_width_ratio[row_index] = times\n    else:\n        times = w/h\n        times = round(times,1)\n        width_height_ratio[row_index] = times\ntrain_size_df['width_height_ratio'] = width_height_ratio\ntrain_size_df['height_width_ratio'] = height_width_ratio","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_size_df['ratio'] = train_size_df['height_width_ratio'] + train_size_df['width_height_ratio']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_size_df['ratio'].max(),train_size_df['ratio'].min(),train_size_df['ratio'].mean(),train_size_df['ratio'].quantile(q=0.25),train_size_df['ratio'].median(),train_size_df['ratio'].quantile(q=0.75)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ratio_1_2 = train_size_df[train_size_df['ratio']<=2]\nprint(ratio_1_2.shape)\nplt.figure(figsize=(20,8))\nax = sns.countplot(ratio_1_2['ratio'])\nplt.xlabel('ratio_1_2')\nplt.title('Number of image per ratio', fontsize=20)\n\nfor p in ax.patches:\n    ax.annotate(f'{p.get_height()}-{p.get_height() * 100 / ratio_1_2.shape[0]:.3f}%',\n            (p.get_x() + p.get_width() / 2., p.get_height()), \n            ha='center', \n            va='center', \n            fontsize=11, \n            color='black',\n            xytext=(0,7), \n            textcoords='offset points')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"ratio_section = [1.0,1.3,1.5,2.0,3.0,4.0,5.0,6.0,7.0,8.0,9.0,10.0,11.0,12.0,13.0,14.0,15.0,16.0,20.0,26.0]\nratio_section_count = np.zeros(len(ratio_section)-1)\nw_ratio_section_count = np.zeros(len(ratio_section)-1)\nh_ratio_section_count = np.zeros(len(ratio_section)-1)\nfor i in range(len(ratio_section_count)):\n    start = ratio_section[i]\n    end = ratio_section[i+1]-0.1\n    ratio_section_count[i] = train_size_df['ratio'].between(start,end).sum()\n    w_ratio_section_count[i] = train_size_df['width_height_ratio'].between(start,end).sum()\n    h_ratio_section_count[i] = train_size_df['height_width_ratio'].between(start,end).sum()\nratio_section_count = ratio_section_count.astype(np.int64)\nw_ratio_section_count = w_ratio_section_count.astype(np.int64)\nh_ratio_section_count = h_ratio_section_count.astype(np.int64)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def draw(start,end,column='ratio'):\n    temp_df = train_size_df[train_size_df[column].between(start,end)]\n    temp_labels_count = np.zeros((len(temp_df),1103))\n    num_temp_labels = np.zeros(len(temp_df))\n    for row_index,row in enumerate(temp_df['attribute_ids']):\n        ids = row.split(' ')\n        num_temp_labels[row_index] = len(ids)\n        for id_index in ids:\n            temp_labels_count[row_index,int(id_index)] = 1\n    label_sum = np.sum(temp_labels_count, axis=0)\n    attributes_sequence = label_sum.argsort()[::-1]\n    label_names = pd.read_csv('../input/labels.csv')\n    label_names = label_names['attribute_name']\n    attributes_labels = [label_names[x] for x in attributes_sequence]\n    attributes_counts = [label_sum[x] for x in attributes_sequence]\n    plt.figure(figsize=(20,2))\n\n    plt.subplot()\n    ax1 = sns.barplot(y=attributes_labels[:5], x=attributes_counts[:5], orient=\"h\")\n    plt.title(f'Label Counts between {start} and {end} (Top 5)',fontsize=15)\n    plt.xlim((0, max(attributes_counts)*1.15))\n    plt.yticks(fontsize=15)\n\n    for p in ax1.patches:\n        ax1.annotate(f'{int(p.get_width())}-{p.get_width() * 100 / temp_df.shape[0]:.2f}%',\n                    (p.get_width(), p.get_y() + p.get_height() / 2.), \n                    ha='left', \n                    va='center', \n                    fontsize=10, \n                    color='black',\n                    xytext=(7,0), \n                    textcoords='offset points')\n    plt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for ratio in [1.0,1.1,1.2,1.3,1.4,1.5,1.6,1.7,1.8,1.9,2.0,2.1,2.2]:\n    draw(ratio,ratio)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for ratio in [1.0,1.1,1.2,1.3,1.4,1.5,1.6,1.7,1.8,1.9,2.0,2.1,2.2]:\n    draw(ratio,ratio,'height_width_ratio')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for ratio in [1.0,1.1,1.2,1.3,1.4,1.5,1.6,1.7,1.8,1.9,2.0,2.1,2.2]:\n    draw(ratio,ratio,'width_height_ratio')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_size_df.to_csv('train_size_df.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_hsv_df = pd.read_csv('../input/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gray_img = np.zeros(len(train_hsv_df))\nh_list = np.zeros(len(train_hsv_df))\ns_list = np.zeros(len(train_hsv_df))\nv_list = np.zeros(len(train_hsv_df))\nfor row,img_name in tqdm(enumerate(train_hsv_df['id'])):\n    img = cv2.imread('../input/train/'+img_name+'.png')\n    hsv_img = cv2.cvtColor(img,cv2.COLOR_BGR2HSV)\n    h,s,v = np.average(hsv_img,axis=(0,1))\n    if h == 0:\n        gray_img[row] = 1\n    h_list[row] = h\n    s_list[row] = s\n    v_list[row] = v\ntrain_hsv_df['gray_img'] = gray_img\ntrain_hsv_df['h'] = h_list\ntrain_hsv_df['s'] = s_list\ntrain_hsv_df['v'] = v_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_hsv_df.to_csv('train_hsv_df.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gray_df = train_hsv_df[train_hsv_df['gray_img']==1]\nprint(gray_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gray_labels = np.zeros((len(gray_df),1103))\nfor row_index,row in enumerate(gray_df['attribute_ids']):\n    for label in row.split(' '):\n        gray_labels[row_index,int(label)] = 1\ntrain_labels = np.zeros((len(train_hsv_df),1103))\nfor row_index,row in enumerate(train_hsv_df['attribute_ids']):\n    for label in row.split(' '):\n        train_labels[row_index,int(label)] = 1","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"gray_sums = gray_labels.sum(axis=0)\ntrain_sums = train_labels.sum(axis=0)\npercentage = gray_sums/train_sums\n\nlabel_names = pd.read_csv('../input/labels.csv')\nlabel_names = label_names['attribute_name']\nnew_img_df = pd.DataFrame(index=label_names)\nnew_img_df['gray_count'] = gray_sums\nnew_img_df['count'] = train_sums\nnew_img_df['percentage'] = percentage\nnew_img_df.head(5)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_img_df.to_csv('new_img_df.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_img_df.sort_values(['percentage'],ascending=False).iloc[:30]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_img_df.sort_values(['gray_count','percentage'],ascending=False).iloc[:30]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"markets = []\nmarkets_index = []\nfor i,name in enumerate(label_names):\n    if name.endswith('market'):\n        print(name)\n        markets.append(name)\n        markets_index.append(i)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"new_img_df.loc[markets]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"markets_index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in np.random.choice(np.where(train_labels[:,134]==1)[0],3):\n    img_path = '../input/train/' + train_hsv_df.iloc[i]['id'] + '.png'\n    img = Image.open(img_path)\n    plt.imshow(img)\n    plt.show()\n    print([label_names[int(i)] for i in train_hsv_df.iloc[i]['attribute_ids'].split(' ')])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in np.random.choice(np.where(train_labels[:,135]==1)[0],3):\n    img_path = '../input/train/' + train_hsv_df.iloc[i]['id'] + '.png'\n    img = Image.open(img_path)\n    plt.imshow(img)\n    plt.show()\n    print([label_names[int(i)] for i in train_hsv_df.iloc[i]['attribute_ids'].split(' ')])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in np.random.choice(np.where(train_labels[:,142]==1)[0],2):\n    img_path = '../input/train/' + train_hsv_df.iloc[i]['id'] + '.png'\n    img = Image.open(img_path)\n    plt.imshow(img)\n    plt.show()\n    print([label_names[int(i)] for i in train_hsv_df.iloc[i]['attribute_ids'].split(' ')])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def mask2(img_path):\n    image=cv2.imread(img_path)\n    img=cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    hsv_image = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n    hsv_image2 = cv2.cvtColor(img, cv2.COLOR_RGB2HSV)\n    h,s,v = cv2.split(hsv_image)\n    lower_blue=np.array([20,30,0])\n    upper_blue=np.array([160,255,255])\n    lower_blue2=np.array([0,30,0])\n    upper_blue2=np.array([255,255,255])\n    mask=cv2.inRange(hsv_image,lower_blue,upper_blue)\n    mask2=cv2.inRange(hsv_image2,lower_blue2,upper_blue2)\n    use_mask = False\n    if np.average(h)<20 or np.average(s)<30 or np.average(v)<140:\n        res = img\n        res2 = cv2.bitwise_and(img,img,mask=mask)\n        res3 = cv2.bitwise_and(img,img,mask=mask2)\n        res4 = (res2+res3)//2\n        res4[np.where(res4>255)]=255\n    else:\n        res = cv2.bitwise_and(img,img,mask=mask)\n        res2 = cv2.bitwise_and(img,img,mask=mask)\n        res3 = cv2.bitwise_and(img,img,mask=mask2)\n        res4 = (res2+res3)//2\n        res4[np.where(res4>255)]=255\n        use_mask = True\n    plt.subplot(171),plt.imshow(img),plt.title('ORIGINAL')\n    plt.subplot(172),plt.imshow(mask),plt.title('Mask1')\n    plt.subplot(173),plt.imshow(mask2),plt.title('Mask2')\n    plt.subplot(174),plt.imshow(res),plt.title('use_mask' if use_mask else 'original')\n    plt.subplot(175),plt.imshow(res2),plt.title('use_mask_BGR')\n    plt.subplot(176),plt.imshow(res3),plt.title('use_mask_RGB')\n    plt.subplot(177),plt.imshow(res4),plt.title('res4')\n    plt.show()\n    h,s,v=np.average(hsv_image,axis=(0,1))\n    print(h, s, v)\n    h2,s2,v2=np.average(hsv_image2,axis=(0,1))\n    print(h2, s2, v2)\n    if h == 0 and s == 0:\n        print(np.average(img,axis=(0,1)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in np.random.choice(range(len(train_hsv_df)),3):\n    plt.figure(figsize=(20,20))\n    img_id = train_hsv_df.iloc[i]['id']\n    path = '../input/train/'+img_id+'.png'\n    mask2(path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}