{"cells":[
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": true
  },
  "outputs": [],
  "source": "%matplotlib inline"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": "#\n# This script creates a plot with the 10 most used ingredients.\n#\n# The original recipe, contained in the 'ingredients' column, is cleaned as follow:\n#\n# - to lowecase\n# - replacing symbols\n# - removing digits\n# - stemming the words using the WordNetLemmatizer\n#\n# The ingredients should be cleaned mote, making 'low fat mozzarella' and \n# 'reduced fat mozzarella' the same ingredient. Ideas are welcome.\n# \n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom nltk.stem import WordNetLemmatizer\nfrom collections import Counter\n\n# Reading the data\ntrain = pd.read_json('../input/train.json')\n\nstemmer = WordNetLemmatizer()\n#cachedStopWords = stopwords.words(\"english\")\n\n# Auxiliar function for cleaning\ndef clean_recipe(recipe):\n    # To lowercase\n    recipe = [ str.lower(i) for i in recipe ]\n\n    # Remove some special characters\n    # Individuals replace have a very good performance\n    # http://stackoverflow.com/a/27086669/670873\n    def replacing(i):\n        i = i.replace('&', '').replace('(', '').replace(')','')\n        i = i.replace('\\'', '').replace('\\\\', '').replace(',','')\n        i = i.replace('.', '').replace('%', '').replace('/','')\n        i = i.replace('\"', '')\n        \n        return i\n    \n    # Replacing characters\n    recipe = [ replacing(i) for i in recipe ]\n    \n    # Remove digits\n    recipe = [ i for i in recipe if not i.isdigit() ]\n    \n    # Stem ingredients\n    recipe = [ stemmer.lemmatize(i) for i in recipe ]\n    \n    return recipe\n\n# The number of times each ingredient is used is stored in the 'sumbags' dictionary\nbags_of_words = [ Counter(clean_recipe(recipe)) for recipe in train.ingredients ]\nsumbags = sum(bags_of_words, Counter())\n\n# Finally, plot the 10 most used ingredients\nplt.style.use(u'ggplot')\nfig = pd.DataFrame(sumbags, index=[0]).transpose()[0].sort(ascending=False, inplace=False)[:20].plot(kind='barh')\nfig.invert_yaxis()\nfig = fig.get_figure()\nfig.tight_layout()\nfig.savefig('10_most_used_ingredients.jpg')\n"
 },
 {
  "cell_type": "code",
  "execution_count": null,
  "metadata": {
   "collapsed": false
  },
  "outputs": [],
  "source": ""
 }
],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}}, "nbformat": 4, "nbformat_minor": 0}