{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.\n\n\n%matplotlib inline\nimport cv2\nfrom scipy.stats import itemfreq\nimport matplotlib\nimport matplotlib.pyplot as plt\n\nimport seaborn as sns # visualizations\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"840003a44c6882a988d9cd6263b4155c86b11ca0"},"cell_type":"markdown","source":"# This Notebook is an exploration of the [\"Painter By Numbers\" dataset](https://www.kaggle.com/c/painter-by-numbers)\n\n# Observations\n1. This dataset has a lot of images\n\n# Questions / Hypotheses\n1. Is it possible to identify the artist of a painting by features related to the painting? \n  * Would the following features be helpful?\n    * color palette\n    * dominant color\n    * etc ..."},{"metadata":{"trusted":true,"_uuid":"801a55ab767b7930a64b23a6be4c0686b816b453"},"cell_type":"code","source":"df = pd.read_csv(\"../input/train_info.csv\")\ndf[df['title'].notnull()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a7728aca563bd855af01ff780272840a1c13ddf","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# https://www.kaggle.com/getting-started/39426\n!pip install --upgrade pip\n!pip install webcolors","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a76e858d5b394ac1183e69fd443fcbc43d4d6f6","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# https://stackoverflow.com/a/9694246/5411712\nimport webcolors","execution_count":null,"outputs":[]},{"metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"trusted":true,"_uuid":"7b0a8c9df65113bb0ff12e901817ecabf4d4a4d2"},"cell_type":"code","source":"# This code helps identify color names\n\n# https://stackoverflow.com/a/9694246/5411712\ndef closest_colour(requested_colour):\n    min_colours = {}\n    for key, name in webcolors.css3_hex_to_names.items():\n        r_c, g_c, b_c = webcolors.hex_to_rgb(key)\n        rd = (r_c - requested_colour[0]) ** 2\n        gd = (g_c - requested_colour[1]) ** 2\n        bd = (b_c - requested_colour[2]) ** 2\n        min_colours[(rd + gd + bd)] = name\n    return min_colours[min(min_colours.keys())]\n\ndef get_colour_name(requested_colour):\n    try:\n        closest_name = actual_name = webcolors.rgb_to_name(requested_colour)\n    except ValueError:\n        closest_name = closest_colour(requested_colour)\n        actual_name = None\n    return actual_name, closest_name","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bddda91aeb0eaa5c21ea008bfde08237d5f01f85"},"cell_type":"markdown","source":"# Here is 1 image from the training set"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"img = cv2.imread('../input/train_2/2.jpg')\nimg = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\nplt.imshow(img)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5694a6548a9c1f5649e3facaddf91f568f90c9f1"},"cell_type":"markdown","source":"# Let's find the ***Color Palette*** of the image, and the ***Dominiant Color***\nResource: https://stackoverflow.com/q/43111029/5411712"},{"metadata":{"trusted":true,"_uuid":"48e6c3bc905b3285569bfa2b64103d312d89f0f6","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# average_color = [img[:, :, i].mean() for i in range(img.shape[-1])]\n#===============\narr = np.float32(img)\npixels = arr.reshape((-1, 3))\n\nn_colors = 5\ncriteria = (cv2.TERM_CRITERIA_EPS + cv2.TERM_CRITERIA_MAX_ITER, 200, .1)\nflags = cv2.KMEANS_RANDOM_CENTERS\n_, labels, centroids = cv2.kmeans(pixels, n_colors, None, criteria, 10, flags)\n\npalette = np.uint8(centroids)\nquantized = palette[labels.flatten()]\nquantized = quantized.reshape(img.shape)\n#===============\ndominant_color = palette[np.argmax(itemfreq(labels)[:, -1])]\n# `itemfreq` is deprecated and will be removed in a future version. Use instead `np.unique(..., return_counts=True)`\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5aebfe8c28f8553bec1ca598f0c8505e68abbb18","_kg_hide-input":true},"cell_type":"code","source":"actual_name, closest_name = get_colour_name(dominant_color)\n\nprint(\"palette = \")\npalette_names = []\nfor color in palette:\n    meh, name = get_colour_name(color)\n    palette_names.append(name)\n    print(str(color) + \" \" + str(name))\nprint('\\n\\n')\n\ndc = dominant_color\n\nprint(\"dominant_color (rgb)\\t=\\t\" + str(dominant_color))\nprint(\"dominant_color (name)\\t=\\t\" + str(closest_name))\nprint(\"dominant_color (hex)\\t=\\t\" + str('#%02x%02x%02x' % (dc[0], dc[1], dc[2])))\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"296d3e33060d56c0e52b30fc774a11909a978994"},"cell_type":"markdown","source":"# SO [***goldenrod***](https://www.google.com/search?q=goldenrod+color) is the dominant color of the image ```../input/train_2/2.jpg```\n\n#### a limitation of this technique/code is that ```#d2d43b``` is not necessarily the same as ```goldenrod```, but it's close enough\n#### maybe just the hex code is enough/better, but using color names would help reduce the feature size. "},{"metadata":{"_kg_hide-input":true,"trusted":true,"_uuid":"2c4d83050c83486041e9625bb02ff24741774bd2"},"cell_type":"code","source":"plt.imshow(img)\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cbeff94d51690a0e0a28f69f13a685577c1e759f"},"cell_type":"markdown","source":"# Here are all the image files and their features\n### * displaying just the first 10 with ```df.head(10)```"},{"metadata":{"trusted":true,"_uuid":"42bde364962e97487f823933f81f96fd27478ff8"},"cell_type":"code","source":"df = pd.read_csv(\"../input/train_info.csv\")\ndf.head(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4979757fe3ecc5d5551f3dcaccb1d78b0dc7836d"},"cell_type":"markdown","source":"# Let's look at the features related to ```../input/train_2/2.jpg```"},{"metadata":{"trusted":true,"_uuid":"93d04dbbc95b96c7cdadd24d2c33587664994df7"},"cell_type":"code","source":"df[(df['filename'] == '2.jpg')]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d817934f344a6aeb300d659afa8dcd0331f1771f"},"cell_type":"markdown","source":"# just two lists of all the ***unique*** ```styles``` and ```genres```"},{"metadata":{"_kg_hide-output":false,"trusted":true,"_uuid":"03d0258e387aa2102787e790d6826159fa0810f4","_kg_hide-input":true},"cell_type":"code","source":"#sns.boxplot(x='style', y='idk', data=df)\n\ndf['style']\n\nstyles = []\n\nfor s in df['style']:\n    if (s not in styles):\n        styles += [s]\n\nprint(\"\\nSTYLES\\n\")\nprint(styles)\n\ngenres = []\n\nfor g in df['genre']:\n    if (g not in genres):\n        genres += [g]\n\nprint(\"\\nGENRES\\n\")\nprint(genres)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f80c66c006da0ca4248f2550d740c7e1473edce9"},"cell_type":"markdown","source":"# The artist names are hashed... werid, but how many artists are there? Also how many images are there per artist? "},{"metadata":{"trusted":true,"_uuid":"aba282b678afa92ae628847e549cc8d0f82e805b","_kg_hide-input":true},"cell_type":"code","source":"artists = {} # holds artist hash & the count\nfor a in df['artist']:\n    if (a not in artists):\n        artists[a] = 1\n    else:\n        artists[a] += 1\n\nprint(\"\\nHow many artists? \\n\")\nprint(len(artists))\n\n# convert unique hashes to unique numbers \n# conversion helps matplotlib actually plot\nnew_dict = {} \ni = 1\nfor each in artists:\n    new_dict[i] = artists[each]\n    i += 1\n\nlists = sorted(new_dict.items(), key=lambda kv: kv[1], reverse=True) # sorted by value, return a list of tuples\nx, y = zip(*lists) # unpack a list of pairs into two tuples\n\nfig_size = plt.rcParams[\"figure.figsize\"]\nfig_size[0] = 12\nfig_size[1] = 9\nplt.rcParams[\"figure.figsize\"] = fig_size\n\nplt.plot(y, x, 'r.')\nplt.title(s='Plot of Number of Images vs Artist ID')\nplt.ylabel(s='Artist ID (not hash but is unique number related to the hash)')\nplt.xlabel(s='Number of Images')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"15b5982390e3e4af4ca423dcb442ec830d733b2e"},"cell_type":"markdown","source":"# How many artists have over 300 images? "},{"metadata":{"trusted":true,"_uuid":"0fe9c71d34b4ce76a2b2f334b9179b249c6b3f99","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"over_200 = 0\nover_300 = 0\nover_400 = 0\nunder_100 = 0\nunder_50 = 0\nunder_25 = 0\nunder_5 = 0\nunder_1 = 0\n\nfor a in artists:\n    over_200 = over_200 + 1 if artists[a] >= 200 else over_200\n    over_300 = over_300 + 1 if artists[a] >= 300 else over_300\n    over_400 = over_400 + 1 if artists[a] >= 400 else over_400\n    under_100 = under_100 + 1 if artists[a] <= 100 else under_100\n    under_50 = under_50 + 1 if artists[a] <= 50 else under_50\n    under_25 = under_25 + 1 if artists[a] <= 25 else under_25\n    under_5 = under_5 + 1 if artists[a] <= 5 else under_5\n    under_1 = under_1 + 1 if artists[a] <= 1 else under_1\n\nprint(\"OVER?\")\nprint(\"over_200 = \" + str(over_200))\nprint(\"over_300 = \" + str(over_300))\nprint(\"over_400 = \" + str(over_400))\n\nprint(\"UNDER?\")\nprint(\"under_100 = \" + str(under_100))\nprint(\"under_50 = \" + str(under_50))\nprint(\"under_25 = \" + str(under_25))\nprint(\"under_5 = \" + str(under_5))\nprint(\"under_1 = \" + str(under_1))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"11fcc60c633e6295e67f53fe9094bcc71317d32c"},"cell_type":"markdown","source":"# some stats plots to show the same info"},{"metadata":{"trusted":true,"_uuid":"246aed8464f2a4ef91d43bf1d56e8cb848123873"},"cell_type":"code","source":"# https://stackoverflow.com/a/20026448/5411712\nimport scipy.stats as stats\nfit = stats.norm.pdf(y, np.mean(y), np.std(y))  #this is a fitting indeed\nplt.plot(y,fit,'-o')\nplt.hist(y, normed=True)\nplt.show()\n\n# https://stackoverflow.com/a/15419072/5411712\nvalues, base = np.histogram(y, bins=40)\ncumulative = np.cumsum(values)\nplt.plot(base[:-1], cumulative, '-o', c='blue')\nplt.plot(base[:-1], len(y)-cumulative, '-o', c='green')\nplt.hist(y, color='orange')\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}